From a04b9ef9c5807c3937429e93fd2166aebb347518 Mon Sep 17 00:00:00 2001
From: SZCJW <792430652@qq.com>
Date: Thu, 26 Mar 2026 20:50:00 +0800
Subject: [PATCH] x
---
.github/workflows/build-and-publish.yml | 98 +
.gitignore | 1 +
.idea/.gitignore | 5 +
.idea/amd-r9700-vllm-toolboxes.iml | 12 +
.idea/inspectionProfiles/Project_Default.xml | 14 +
.../inspectionProfiles/profiles_settings.xml | 6 +
.idea/misc.xml | 7 +
.idea/modules.xml | 8 +
.idea/vcs.xml | 6 +
Dockerfile | 164 +
README.md | 209 +
TUNING.md | 108 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 16815 ++++++++++++++++
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...HatAI_Qwen3-14B-FP8-dynamic_tp2_server.log | 432 +
..._Qwen3-14B-FP8-dynamic_tp2_throughput.json | 7 +
...12b-it-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...12b-it-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
..._gemma-3-12b-it-FP8-dynamic_tp1_server.log | 1031 +
...a-3-12b-it-FP8-dynamic_tp1_throughput.json | 7 +
...12b-it-FP8-dynamic_tp2_qps1.0_latency.json | 4 +
...12b-it-FP8-dynamic_tp2_qps4.0_latency.json | 4 +
..._gemma-3-12b-it-FP8-dynamic_tp2_server.log | 1053 +
...a-3-12b-it-FP8-dynamic_tp2_throughput.json | 7 +
...27b-it-FP8-dynamic_tp2_qps1.0_latency.json | 4 +
...27b-it-FP8-dynamic_tp2_qps4.0_latency.json | 4 +
..._gemma-3-27b-it-FP8-dynamic_tp2_server.log | 2570 +++
...a-3-27b-it-FP8-dynamic_tp2_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp1_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp1_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log | 1038 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp2_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp2_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log | 846 +
...A3B-Instruct-GPTQ-4bit_tp2_throughput.json | 7 +
...t-80B-A3B-Instruct-AWQ-4bit_tp2_server.log | 941 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1027 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp2_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp2_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp2_server.log | 1034 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1033 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
.../openai_gpt-oss-20b_tp2_server.log | 437 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 1042 +
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1040 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1045 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 1047 +
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...12b-it-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...12b-it-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
..._gemma-3-12b-it-FP8-dynamic_tp1_server.log | 1048 +
...a-3-12b-it-FP8-dynamic_tp1_throughput.json | 7 +
...12b-it-FP8-dynamic_tp2_qps1.0_latency.json | 4 +
...12b-it-FP8-dynamic_tp2_qps4.0_latency.json | 4 +
..._gemma-3-12b-it-FP8-dynamic_tp2_server.log | 1074 +
...a-3-12b-it-FP8-dynamic_tp2_throughput.json | 7 +
...27b-it-FP8-dynamic_tp2_qps1.0_latency.json | 4 +
...27b-it-FP8-dynamic_tp2_qps4.0_latency.json | 4 +
..._gemma-3-27b-it-FP8-dynamic_tp2_server.log | 1092 +
...a-3-27b-it-FP8-dynamic_tp2_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp1_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp1_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log | 1054 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp2_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp2_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log | 1076 +
...A3B-Instruct-GPTQ-4bit_tp2_throughput.json | 7 +
...-Instruct-AWQ-4bit_tp2_qps1.0_latency.json | 4 +
...-Instruct-AWQ-4bit_tp2_qps4.0_latency.json | 4 +
...t-80B-A3B-Instruct-AWQ-4bit_tp2_server.log | 1128 ++
...-A3B-Instruct-AWQ-4bit_tp2_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1043 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp2_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp2_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp2_server.log | 1065 +
...-Llama-3.1-8B-Instruct_tp2_throughput.json | 7 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1050 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
...openai_gpt-oss-20b_tp2_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp2_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp2_server.log | 328 +
.../openai_gpt-oss-20b_tp2_throughput.json | 7 +
...a-3.1-8B-Instruct-FP8-block_tp1_server.log | 95 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 1027 +
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp1_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp1_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log | 1026 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1023 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1022 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 1024 +
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp1_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp1_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log | 1026 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1023 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1022 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 1032 +
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp1_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp1_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log | 1029 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1027 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1030 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 1024 +
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp1_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp1_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log | 1026 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1022 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1023 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
...n3-14B-FP8-dynamic_tp1_qps1.0_latency.json | 4 +
...n3-14B-FP8-dynamic_tp1_qps4.0_latency.json | 4 +
...HatAI_Qwen3-14B-FP8-dynamic_tp1_server.log | 1025 +
..._Qwen3-14B-FP8-dynamic_tp1_throughput.json | 7 +
...Instruct-GPTQ-4bit_tp1_qps1.0_latency.json | 4 +
...Instruct-GPTQ-4bit_tp1_qps4.0_latency.json | 4 +
...-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log | 1025 +
...A3B-Instruct-GPTQ-4bit_tp1_throughput.json | 7 +
...ma-3.1-8B-Instruct_tp1_qps1.0_latency.json | 4 +
...ma-3.1-8B-Instruct_tp1_qps4.0_latency.json | 4 +
..._Meta-Llama-3.1-8B-Instruct_tp1_server.log | 1021 +
...-Llama-3.1-8B-Instruct_tp1_throughput.json | 7 +
...openai_gpt-oss-20b_tp1_qps1.0_latency.json | 4 +
...openai_gpt-oss-20b_tp1_qps4.0_latency.json | 4 +
.../openai_gpt-oss-20b_tp1_server.log | 1022 +
.../openai_gpt-oss-20b_tp1_throughput.json | 7 +
benchmarks/find_max_context.py | 572 +
benchmarks/generate_readme_table.py | 119 +
benchmarks/max_context_results.json | 611 +
benchmarks/run_vllm_bench.py | 356 +
benchmarks/run_vllm_bench_nvidia.py | 392 +
demo.gif | Bin 0 -> 4188469 bytes
docs/assets/index2.css | 401 +
docs/assets/index2.js | 542 +
docs/compare.html | 804 +
docs/comparison_results.json | 54 +
docs/generate_comparison_data.py | 107 +
docs/index.html | 885 +
docs/parse_results.py | 139 +
docs/results.json | 1404 ++
scripts/01-rocm-envs.sh | 4 +
scripts/99-toolbox-banner.sh | 108 +
scripts/generate_models_list.py | 97 +
scripts/start_vllm.py | 358 +
scripts/zz-venv-last.sh | 16 +
208 files changed, 71242 insertions(+)
create mode 100644 .github/workflows/build-and-publish.yml
create mode 100644 .gitignore
create mode 100644 .idea/.gitignore
create mode 100644 .idea/amd-r9700-vllm-toolboxes.iml
create mode 100644 .idea/inspectionProfiles/Project_Default.xml
create mode 100644 .idea/inspectionProfiles/profiles_settings.xml
create mode 100644 .idea/misc.xml
create mode 100644 .idea/modules.xml
create mode 100644 .idea/vcs.xml
create mode 100644 Dockerfile
create mode 100644 README.md
create mode 100644 TUNING.md
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_server.log
create mode 100644 benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/RedHatAI_Llama-3.1-8B-Instruct-FP8-block_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/openai_gpt-oss-20b_tp1_qps1.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/openai_gpt-oss-20b_tp1_qps4.0_latency.json
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/openai_gpt-oss-20b_tp1_server.log
create mode 100644 benchmarks/benchmark_results_nvidia-ada5000/openai_gpt-oss-20b_tp1_throughput.json
create mode 100644 benchmarks/find_max_context.py
create mode 100644 benchmarks/generate_readme_table.py
create mode 100644 benchmarks/max_context_results.json
create mode 100644 benchmarks/run_vllm_bench.py
create mode 100644 benchmarks/run_vllm_bench_nvidia.py
create mode 100644 demo.gif
create mode 100644 docs/assets/index2.css
create mode 100644 docs/assets/index2.js
create mode 100644 docs/compare.html
create mode 100644 docs/comparison_results.json
create mode 100644 docs/generate_comparison_data.py
create mode 100644 docs/index.html
create mode 100644 docs/parse_results.py
create mode 100644 docs/results.json
create mode 100644 scripts/01-rocm-envs.sh
create mode 100644 scripts/99-toolbox-banner.sh
create mode 100644 scripts/generate_models_list.py
create mode 100644 scripts/start_vllm.py
create mode 100644 scripts/zz-venv-last.sh
diff --git a/.github/workflows/build-and-publish.yml b/.github/workflows/build-and-publish.yml
new file mode 100644
index 0000000..5778bd2
--- /dev/null
+++ b/.github/workflows/build-and-publish.yml
@@ -0,0 +1,98 @@
+name: build-and-publish
+
+on:
+ workflow_dispatch:
+ inputs:
+ tag:
+ description: "Image tag to publish (e.g. latest)"
+ required: true
+ default: "latest"
+
+env:
+ IMAGE_REPO: kyuz0/vllm-therock-gfx1201
+ DOCKER_BUILDKIT: "1"
+
+jobs:
+ build:
+ runs-on: ubuntu-latest
+ steps:
+ - name: Put Docker on /mnt
+ run: |
+ set -eux
+ echo '{ "data-root": "/mnt/docker" }' | sudo tee /etc/docker/daemon.json
+ sudo systemctl stop docker
+ sudo rm -rf /var/lib/docker || true
+ sudo mkdir -p /mnt/docker
+ sudo systemctl start docker
+ docker info | grep "Docker Root Dir"
+
+ - name: Checkout
+ uses: actions/checkout@v4
+
+ - name: Free disk space
+ shell: bash
+ run: |
+ set -euxo pipefail
+ echo "Disk BEFORE:"; df -h
+ sudo rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc || true
+ sudo rm -rf /opt/hostedtoolcache/CodeQL /opt/hostedtoolcache/go || true
+ docker system prune -af || true
+ docker builder prune -af || true
+ sudo apt-get clean
+ sudo rm -rf /var/lib/apt/lists/*
+ sudo rm -rf /opt/hostedtoolcache
+ echo "Disk AFTER:"; df -h
+
+ - name: Set up QEMU
+ uses: docker/setup-qemu-action@v3
+
+ - name: BuildKit GC config (8GB cap)
+ run: |
+ cat > /tmp/buildkitd.toml <<'EOF'
+ [worker.oci]
+ gc = true
+ gckeepstorage = 8000 # MB
+ EOF
+
+ - name: Set up Buildx (with GC)
+ uses: docker/setup-buildx-action@v3
+ with:
+ buildkitd-flags: --config /tmp/buildkitd.toml
+
+ - name: Log in to Docker Hub
+ uses: docker/login-action@v3
+ with:
+ username: ${{ secrets.DOCKERHUB_USERNAME }}
+ password: ${{ secrets.DOCKERHUB_TOKEN }}
+
+ - name: Docker meta
+ id: meta
+ uses: docker/metadata-action@v5
+ with:
+ images: docker.io/${{ env.IMAGE_REPO }}
+ tags: |
+ type=raw,value=${{ github.event.inputs.tag }}
+ type=sha
+ type=raw,value={{date 'YYYYMMDD-HHmmss'}}
+ labels: |
+ org.opencontainers.image.source=https://github.com/${{ github.repository }}
+ org.opencontainers.image.revision=${{ github.sha }}
+
+ - name: Build and push
+ id: build
+ uses: docker/build-push-action@v6
+ with:
+ context: .
+ file: ./Dockerfile
+ platforms: linux/amd64
+ push: true
+ tags: ${{ steps.meta.outputs.tags }}
+ labels: ${{ steps.meta.outputs.labels }}
+ provenance: false
+ sbom: false
+ no-cache: true
+
+ - name: Prune buildx cache
+ if: always()
+ run: |
+ docker buildx prune -af --verbose --min-free-space 4gb || true
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000..ed8ebf5
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1 @@
+__pycache__
\ No newline at end of file
diff --git a/.idea/.gitignore b/.idea/.gitignore
new file mode 100644
index 0000000..10b731c
--- /dev/null
+++ b/.idea/.gitignore
@@ -0,0 +1,5 @@
+# 默认忽略的文件
+/shelf/
+/workspace.xml
+# 基于编辑器的 HTTP 客户端请求
+/httpRequests/
diff --git a/.idea/amd-r9700-vllm-toolboxes.iml b/.idea/amd-r9700-vllm-toolboxes.iml
new file mode 100644
index 0000000..2a15c21
--- /dev/null
+++ b/.idea/amd-r9700-vllm-toolboxes.iml
@@ -0,0 +1,12 @@
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/inspectionProfiles/Project_Default.xml b/.idea/inspectionProfiles/Project_Default.xml
new file mode 100644
index 0000000..c677c95
--- /dev/null
+++ b/.idea/inspectionProfiles/Project_Default.xml
@@ -0,0 +1,14 @@
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml
new file mode 100644
index 0000000..105ce2d
--- /dev/null
+++ b/.idea/inspectionProfiles/profiles_settings.xml
@@ -0,0 +1,6 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/misc.xml b/.idea/misc.xml
new file mode 100644
index 0000000..eb2d36c
--- /dev/null
+++ b/.idea/misc.xml
@@ -0,0 +1,7 @@
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/modules.xml b/.idea/modules.xml
new file mode 100644
index 0000000..0159947
--- /dev/null
+++ b/.idea/modules.xml
@@ -0,0 +1,8 @@
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/vcs.xml b/.idea/vcs.xml
new file mode 100644
index 0000000..94a25f7
--- /dev/null
+++ b/.idea/vcs.xml
@@ -0,0 +1,6 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/Dockerfile b/Dockerfile
new file mode 100644
index 0000000..8853520
--- /dev/null
+++ b/Dockerfile
@@ -0,0 +1,164 @@
+FROM registry.fedoraproject.org/fedora:43
+
+# 1. System Base & Build Tools
+# Added 'gperftools-libs' for tcmalloc (fixes double-free)
+RUN dnf -y install --setopt=install_weak_deps=False --nodocs \
+ python3.13 python3.13-devel git rsync libatomic bash ca-certificates curl \
+ gcc gcc-c++ binutils make ffmpeg-free \
+ cmake ninja-build aria2c tar xz vim nano \
+ libdrm-devel zlib-devel openssl-devel jq \
+ numactl-devel gperftools-libs dialog procps-ng \
+ && dnf clean all && rm -rf /var/cache/dnf/*
+
+# 2. Install "TheRock" ROCm SDK (Tarball Method)
+WORKDIR /tmp
+ARG ROCM_MAJOR_VER=7
+ARG GFX=gfx120X-all
+RUN set -euo pipefail; \
+ BASE="https://therock-nightly-tarball.s3.amazonaws.com"; \
+ PREFIX="therock-dist-linux-${GFX}-${ROCM_MAJOR_VER}"; \
+ KEY="$(curl -s "${BASE}?list-type=2&prefix=${PREFIX}" \
+ | tr '<' '\n' \
+ | grep -o "therock-dist-linux-${GFX}-${ROCM_MAJOR_VER}\..*\.tar\.gz" \
+ | sort -V | tail -n1)"; \
+ echo "Downloading Latest Tarball: ${KEY}"; \
+ aria2c -x 16 -s 16 -j 16 --file-allocation=none "${BASE}/${KEY}" -o therock.tar.gz; \
+ mkdir -p /opt/rocm; \
+ tar xzf therock.tar.gz -C /opt/rocm --strip-components=1; \
+ rm therock.tar.gz
+
+# 3. Configure Global ROCm Environment
+# We add LD_PRELOAD for tcmalloc here to fix the shutdown crash
+RUN export ROCM_PATH=/opt/rocm && \
+ BITCODE_PATH=$(find /opt/rocm -type d -name bitcode -print -quit) && \
+ printf '%s\n' \
+ "export ROCM_PATH=/opt/rocm" \
+ "export HIP_PLATFORM=amd" \
+ "export HIP_PATH=/opt/rocm" \
+ "export HIP_CLANG_PATH=/opt/rocm/llvm/bin" \
+ "export HIP_DEVICE_LIB_PATH=$BITCODE_PATH" \
+ "export PATH=$ROCM_PATH/bin:$ROCM_PATH/llvm/bin:\$PATH" \
+ "export LD_LIBRARY_PATH=$ROCM_PATH/lib:$ROCM_PATH/lib64:$ROCM_PATH/llvm/lib:\$LD_LIBRARY_PATH" \
+ "export ROCBLAS_USE_HIPBLASLT=1" \
+ "export TORCH_ROCM_AOTRITON_ENABLE_EXPERIMENTAL=1" \
+ "export VLLM_TARGET_DEVICE=rocm" \
+ "export HIP_FORCE_DEV_KERNARG=1" \
+ "export RAY_EXPERIMENTAL_NOSET_ROCR_VISIBLE_DEVICES=1" \
+ "export LD_PRELOAD=/usr/lib64/libtcmalloc_minimal.so.4" \
+ > /etc/profile.d/rocm-sdk.sh && \
+ chmod 0644 /etc/profile.d/rocm-sdk.sh
+
+# 4. Python Venv Setup
+RUN /usr/bin/python3.13 -m venv /opt/venv
+ENV VIRTUAL_ENV=/opt/venv
+ENV PATH=/opt/venv/bin:$PATH
+ENV PIP_NO_CACHE_DIR=1
+RUN printf 'source /opt/venv/bin/activate\n' > /etc/profile.d/venv.sh
+RUN python -m pip install --upgrade pip wheel packaging "setuptools<80.0.0"
+
+# 5. Install PyTorch (TheRock Nightly)
+RUN python -m pip install \
+ --index-url https://rocm.nightlies.amd.com/v2-staging/gfx120X-all/ \
+ --pre torch torchaudio torchvision
+
+# Flash-Attention
+WORKDIR /opt
+ENV FLASH_ATTENTION_TRITON_AMD_ENABLE="TRUE"
+
+RUN git clone https://github.com/ROCm/flash-attention.git &&\
+ cd flash-attention &&\
+ git checkout main_perf &&\
+ python setup.py install && \
+ cd /opt && rm -rf /opt/flash-attention
+
+# 6. Clone vLLM
+RUN git clone https://github.com/vllm-project/vllm.git /opt/vllm
+WORKDIR /opt/vllm
+
+# --- PATCHING ---
+# vLLM relies on 'amdsmi' to detect AMD GPUs. If it's missing or fails (common in containers),
+# vLLM falls back to CPU. We patch it to force ROCm detection.
+RUN echo "import sys, re" > patch_vllm.py && \
+ echo "from pathlib import Path" >> patch_vllm.py && \
+ # Patch 1: __init__.py - Force is_rocm=True and bypass amdsmi checks
+ echo "p = Path('vllm/platforms/__init__.py')" >> patch_vllm.py && \
+ echo "txt = p.read_text()" >> patch_vllm.py && \
+ echo "txt = txt.replace('import amdsmi', '# import amdsmi')" >> patch_vllm.py && \
+ echo "txt = re.sub(r'is_rocm = .*', 'is_rocm = True', txt)" >> patch_vllm.py && \
+ echo "txt = re.sub(r'if len\(amdsmi\.amdsmi_get_processor_handles\(\)\) > 0:', 'if True:', txt)" >> patch_vllm.py && \
+ echo "txt = txt.replace('amdsmi.amdsmi_init()', 'pass')" >> patch_vllm.py && \
+ echo "txt = txt.replace('amdsmi.amdsmi_shut_down()', 'pass')" >> patch_vllm.py && \
+ echo "p.write_text(txt)" >> patch_vllm.py && \
+ # Patch 2: rocm.py - Mock amdsmi and force device name
+ echo "p = Path('vllm/platforms/rocm.py')" >> patch_vllm.py && \
+ echo "txt = p.read_text()" >> patch_vllm.py && \
+ echo "header = 'import sys\nfrom unittest.mock import MagicMock\nsys.modules[\"amdsmi\"] = MagicMock()\n'" >> patch_vllm.py && \
+ echo "txt = header + txt" >> patch_vllm.py && \
+ echo "txt = re.sub(r'device_type = .*', 'device_type = \"rocm\"', txt)" >> patch_vllm.py && \
+ echo "txt = re.sub(r'device_name = .*', 'device_name = \"gfx1201\"', txt)" >> patch_vllm.py && \
+ echo "txt += '\n def get_device_name(self, device_id: int = 0) -> str:\n return \"AMD-gfx1201\"\n'" >> patch_vllm.py && \
+ echo "p.write_text(txt)" >> patch_vllm.py && \
+ echo "print('Successfully patched vLLM for R9700')" >> patch_vllm.py && \
+ python patch_vllm.py
+
+# 7. Build vLLM (Wheel Method) with CLANG Host Compiler
+RUN python -m pip install --upgrade cmake ninja packaging wheel numpy "setuptools-scm>=8" "setuptools<80.0.0" scikit-build-core pybind11
+ENV ROCM_HOME="/opt/rocm"
+ENV HIP_PATH="/opt/rocm"
+ENV VLLM_TARGET_DEVICE="rocm"
+ENV PYTORCH_ROCM_ARCH="gfx1201"
+ENV HIP_ARCHITECTURES="gfx1201"
+ENV AMDGPU_TARGETS="gfx1201"
+ENV MAX_JOBS="4"
+
+# --- FIX FOR SEGFAULT ---
+# We force the Host Compiler (CC/CXX) to be the ROCm Clang, not Fedora GCC.
+# This aligns the ABI of the compiled vLLM extensions with PyTorch.
+ENV CC="/opt/rocm/llvm/bin/clang"
+ENV CXX="/opt/rocm/llvm/bin/clang++"
+
+RUN export HIP_DEVICE_LIB_PATH=$(find /opt/rocm -type d -name bitcode -print -quit) && \
+ echo "Compiling with Bitcode: $HIP_DEVICE_LIB_PATH" && \
+ export CMAKE_PREFIX_PATH="/opt/venv/lib64/python3.13/site-packages/torch/share/cmake:/opt/rocm" && \
+ export CMAKE_ARGS="-DROCM_PATH=/opt/rocm -DHIP_PATH=/opt/rocm -DAMDGPU_TARGETS=gfx1201 -DHIP_ARCHITECTURES=gfx1201 -DCMAKE_PREFIX_PATH=/opt/venv/lib64/python3.13/site-packages/torch/share/cmake:/opt/rocm" && \
+ python -m pip wheel --no-build-isolation --no-deps -w /tmp/dist -v . && \
+ python -m pip install /tmp/dist/*.whl
+
+# --- bitsandbytes (ROCm) ---
+WORKDIR /opt
+RUN git clone -b rocm_enabled_multi_backend https://github.com/ROCm/bitsandbytes.git
+WORKDIR /opt/bitsandbytes
+
+# Explicitly set HIP_PLATFORM (Docker ENV, not /etc/profile)
+ENV HIP_PLATFORM="amd"
+ENV CMAKE_PREFIX_PATH="/opt/rocm"
+
+# Force CMake to use the System ROCm Compiler (/opt/rocm/llvm/bin/clang++)
+RUN cmake -S . \
+ -DGPU_TARGETS="gfx1201" \
+ -DBNB_ROCM_ARCH="gfx1201" \
+ -DCOMPUTE_BACKEND=hip \
+ -DCMAKE_HIP_COMPILER=/opt/rocm/llvm/bin/clang++ \
+ -DCMAKE_CXX_COMPILER=/opt/rocm/llvm/bin/clang++ \
+ && \
+ make -j$(nproc) && \
+ python -m pip install --no-cache-dir . --no-build-isolation --no-deps
+
+# 8. Final Cleanup & Runtime
+WORKDIR /opt
+RUN chmod -R a+rwX /opt && \
+ find /opt/venv -type f -name "*.so" -exec strip -s {} + 2>/dev/null || true && \
+ find /opt/venv -type d -name "__pycache__" -prune -exec rm -rf {} + && \
+ rm -rf /root/.cache/pip || true && \
+ dnf clean all && rm -rf /var/cache/dnf/*
+
+COPY scripts/01-rocm-envs.sh /etc/profile.d/01-rocm-envs.sh
+COPY scripts/99-toolbox-banner.sh /etc/profile.d/99-toolbox-banner.sh
+COPY scripts/zz-venv-last.sh /etc/profile.d/zz-venv-last.sh
+COPY scripts/start_vllm.py /usr/local/bin/start-vllm
+COPY benchmarks/max_context_results.json /opt/max_context_results.json
+COPY benchmarks/run_vllm_bench.py /opt/run_vllm_bench.py
+RUN chmod 0644 /etc/profile.d/*.sh && chmod +x /usr/local/bin/start-vllm && chmod 0644 /opt/max_context_results.json
+RUN printf 'ulimit -S -c 0\n' > /etc/profile.d/90-nocoredump.sh && chmod 0644 /etc/profile.d/90-nocoredump.sh
+
+CMD ["/bin/bash"]
diff --git a/README.md b/README.md
new file mode 100644
index 0000000..9f3a1fd
--- /dev/null
+++ b/README.md
@@ -0,0 +1,209 @@
+# AMD Radeon 9700 AI PRO (gfx1201) — vLLM Toolbox/Container
+
+An **fedora-based** Docker/Podman container that is **Toolbx-compatible** (usable as a Fedora toolbox) for serving LLMs with **vLLM** on **AMD Radeon R9700 (gfx1201)**. Built on the TheRock nightly builds for ROCM.
+
+
+
+---
+
+## Table of Contents
+
+* [Tested Models (Benchmarks)](#tested-models-benchmarks)
+* [1) Toolbx vs Docker/Podman](#1-toolbx-vs-dockerpodman)
+* [2) Quickstart — Fedora Toolbx](#2-quickstart--fedora-toolbx)
+* [3) Quickstart — Ubuntu (Distrobox)](#3-quickstart--ubuntu-distrobox)
+* [4) Testing the API](#4-testing-the-api)
+* [5) Use a Web UI for Chatting](#5-use-a-web-ui-for-chatting)
+
+
+## Tested Models (Benchmarks)
+
+View full benchmarks at: [https://kyuz0.github.io/amd-r9700-vllm-toolboxes/](https://kyuz0.github.io/amd-r9700-vllm-toolboxes/)
+
+*Run benchmarks now include a comparison between the default Triton backend and the optional ROCm attention backend.*
+
+
+**Table Key:** Cell values represent `Max Context Length (GPU Memory Utilization)`.
+
+| Model | TP | 1 Req | 4 Reqs | 8 Reqs | 16 Reqs |
+| :--- | :--- | :--- | :--- | :--- | :--- |
+| **`meta-llama/Meta-Llama-3.1-8B-Instruct`** | 1 | 127k (0.98) | 127k (0.98) | 127k (0.98) | 127k (0.98) |
+| | 2 | 105k (0.98) | 105k (0.98) | 105k (0.98) | 105k (0.98) |
+| **`openai/gpt-oss-20b`** | 1 | 131k (0.98) | 131k (0.98) | 131k (0.98) | 131k (0.98) |
+| | 2 | 131k (0.95) | 131k (0.95) | 131k (0.95) | 131k (0.95) |
+| **`RedHatAI/Qwen3-14B-FP8-dynamic`** | 1 | 41k (0.98) | 41k (0.98) | 41k (0.98) | 41k (0.98) |
+| | 2 | 41k (0.95) | 41k (0.95) | 41k (0.95) | 41k (0.95) |
+| **`cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit`** | 1 | 151k (0.98) | 151k (0.98) | 151k (0.98) | 151k (0.98) |
+| | 2 | 262k (0.98) | 262k (0.98) | 262k (0.98) | 262k (0.98) |
+| **`cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit`** | 2 | 156k (0.98) | 156k (0.98) | 156k (0.98) | 156k (0.98) |
+| **`RedHatAI/gemma-3-12b-it-FP8-dynamic`** | 1 | 45k (0.98) | 45k (0.98) | 45k (0.98) | 45k (0.98) |
+| | 2 | 126k (0.98) | 126k (0.98) | 121k (0.95) | 121k (0.95) |
+| **`RedHatAI/gemma-3-27b-it-FP8-dynamic`** | 2 | 60k (0.98) | 60k (0.98) | 60k (0.98) | 60k (0.98) |
+
+### Advanced Tuning
+
+See [TUNING.md](TUNING.md) for a guide on how to enable undervolting and raise the power limit on AMD R9700 cards on Linux to improve performance and efficiency.
+
+### 🆕 Update: Comparison of Attention Backends (Triton vs ROCm)
+
+*Added Support for ROCm Native Attention Backend*
+
+I have added the ability to switch between the default **Triton** backend and the experimental **ROCm** native backend for attention operations. This provides you with more flexibility to optimize for stability or throughput depending on your specific model and workload.
+
+| Backend | Stability | Throughput | Compatibility |
+| :--- | :--- | :--- | :--- |
+| **Triton** (Default) | ✅ **High** | 🔸 Good | Works with all tested models |
+| **ROCm** | ⚠️ **Experimental** | 🚀 **Highest** | May fail with complex architectures |
+
+**Key Differences:**
+- **Triton**: The safe choice. It uses the Triton compiler to generate kernels and is the standard for vLLM on AMD.
+- **ROCm**: Uses composable kernel based attention. In my benchmarks, this often yields higher throughput (tokens/sec) but can be less stable, leading to crashes or "invalid graph" errors on some newer models.
+
+**How to Use:**
+1. **Easy Mode**: Select the backend in the `start-vllm` wizard (Item 5 in the menu).
+2. **Manual Mode**: Export the following environment variables before running `vllm serve`:
+ ```bash
+ export VLLM_V1_USE_PREFILL_DECODE_ATTENTION=1
+ export VLLM_USE_TRITON_FLASH_ATTN=0
+ ```
+
+
+---
+
+## 1) Toolbx vs Docker/Podman
+
+The `kyuz0/vllm-therock-gfx1201:latest` image can be used both as:
+
+* **Fedora Toolbx (recommended for development):** Toolbx shares your **HOME** and user, so models/configs live on the host. Great for iterating quickly while keeping the host clean.
+* **Docker/Podman (recommended for deployment/perf):** Use for running vLLM as a service (host networking, IPC tuning, etc.). Always **mount a host directory** for model weights so they stay outside the container.
+
+---
+
+## 2) Quickstart — Fedora Toolbx
+
+Create a toolbox that exposes the GPU and relaxes seccomp to avoid ROCm syscall issues:
+
+```bash
+toolbox create vllm-r9700 \
+ --image docker.io/kyuz0/vllm-therock-gfx1201:latest \
+ -- --device /dev/dri --device /dev/kfd \
+ --group-add video --group-add render --security-opt seccomp=unconfined
+```
+
+Enter it:
+
+```bash
+toolbox enter vllm-r9700
+```
+
+**Model storage:** Models are downloaded to `~/.cache/huggingface` by default. This directory is shared with the host if you created the toolbox correctly, so downloads persist.
+
+### Serving a Model (Easiest Way)
+
+The toolbox includes a TUI wizard called **`start-vllm`** which includes pre-configured models and handles launch flags. It also allows you to select the experimental **ROCm attention backend**. This is the easiest way to get started.
+
+```bash
+# if your weights live on disk instead of on HuggingFace, point the
+# launcher at the directory that contains the model folders. the
+# script will look for a subdirectory matching the repo ID and use it
+# when launching.
+export LOCAL_MODEL_DIR=/workspace/models
+start-vllm
+
+# you can also just run the CLI yourself and pass the path directly:
+vllm serve /workspace/models/cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit ...
+```
+
+> **Cache note:** vLLM writes compiled kernels to `~/.cache/vllm/`.
+
+---
+
+### 离线/本地模式
+
+如果你希望完全禁用网络访问,只使用本地权重,可以利用 `LOCAL_MODEL_DIR`。
+脚本会在该目录下查找模型子目录;**找不到时会立即报错并退出**,
+不会尝试下载任何内容。适用于没有外网或受限环境的部署。
+
+```bash
+export LOCAL_MODEL_DIR=/workspace/models
+start-vllm # 本地模式,如果模型缺失则失败
+```
+
+### 离线/本地模式
+
+如果你希望完全禁用网络访问,只使用本地权重,可以利用 `LOCAL_MODEL_DIR`。
+脚本会在该目录下查找模型子目录;**找不到时会立即报错并退出**,
+不会尝试下载任何内容。适用于没有外网或受限环境的部署。
+
+```bash
+export LOCAL_MODEL_DIR=/workspace/models
+start-vllm # 本地模式,如果模型缺失则失败
+```
+
+## 3) Quickstart — Ubuntu (Distrobox)
+
+Ubuntu’s toolbox package still breaks GPU access, so use Distrobox instead:
+
+```bash
+distrobox create -n vllm-r9700 \
+ --image docker.io/kyuz0/vllm-therock-gfx1201:latest \
+ --additional-flags "--device /dev/kfd --device /dev/dri --group-add video --group-add render --security-opt seccomp=unconfined"
+
+distrobox enter vllm-r9700
+```
+
+> **Verification:** Run `rocm-smi` to check GPU status.
+
+### Serving a Model
+Same as above, you can use the **`start-vllm`** wizard to launch models easily.
+
+```bash
+start-vllm
+```
+
+---
+
+## 4) Testing the API
+
+Once the server is up, hit the OpenAI‑compatible endpoint:
+
+```bash
+curl -X POST http://localhost:8000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -d '{"model":"Qwen/Qwen2.5-7B-Instruct","messages":[{"role":"user","content":"Hello! Test the performance."}]}'
+```
+
+You should receive a JSON response with a `choices[0].message.content` reply.
+
+If you don't want to bother specifying the model name, you can run this which will query the currently deployed model:
+
+```bash
+MODEL=$(curl -s http://localhost:8000/v1/models | jq -r '.data[0].id') curl -X POST http://localhost:8000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -d "{
+ \"model\": \"$MODEL\",
+ \"messages\":[{\"role\":\"user\",\"content\":\"Hello! Test the performance.\"}]
+ }"
+```
+
+---
+
+## 5) Use a Web UI for Chatting
+
+If vLLM is on a remote server, expose port 8000 via SSH port forwarding:
+
+```bash
+ssh -L 0.0.0.0:8000:localhost:8000
+```
+
+Then, you can start HuggingFace ChatUI like this (on your host):
+
+```bash
+docker run -p 3000:3000 \
+ --add-host=host.docker.internal:host-gateway \
+ -e OPENAI_BASE_URL=http://host.docker.internal:8000/v1 \
+ -e OPENAI_API_KEY=dummy \
+ -v chat-ui-data:/data \
+ ghcr.io/huggingface/chat-ui-db
+```
+
diff --git a/TUNING.md b/TUNING.md
new file mode 100644
index 0000000..185fb04
--- /dev/null
+++ b/TUNING.md
@@ -0,0 +1,108 @@
+# AMD Radeon 9700 AI PRO (Navi 48) Tuning Guide
+
+This guide show you how to enable undervolting and raise the power limit on the Radeon PRO R9700 on Fedora 43.
+
+## 1. Prerequisites (Enable Overclocking)
+
+The `ppfeaturemask` must be set to unlock voltage control.
+
+```bash
+sudo grubby --update-kernel=ALL --args="amdgpu.ppfeaturemask=0xffffffff"
+```
+
+**Reboot your system after running this.**
+
+## 2. Identify Your GPU
+
+You need the **PCI Bus ID** of the target card (e.g., `07:00.0`).
+
+Run this command to list your AMD GPUs:
+
+```bash
+lspci -nn | grep "VGA" | grep "AMD"
+```
+
+*Example output:*
+`07:00.0 VGA compatible controller...`
+*(In this example, your ID is `0000:07:00.0`)*
+
+## 3. Tuning Script
+
+Save this script as `tune_r9700.sh`.
+
+**Important:** Edit the `PCI_ID` variable at the top to match the ID you found in Step 2.
+
+```bash
+#!/bin/bash
+
+# --- CONFIGURATION ---
+# Replace this with your specific PCI ID from 'lspci'
+# Format must be 0000:XX:XX.X
+PCI_ID="0000:07:00.0"
+# ---------------------
+
+# 1. Robustly Find the Card Name (e.g., card1) directly from the PCI Bus
+# We look inside the PCI device's 'drm' folder for a folder starting with 'card' followed only by numbers.
+if [ ! -d "/sys/bus/pci/devices/$PCI_ID/drm" ]; then
+ echo "Error: PCI Device $PCI_ID not found or has no DRM driver attached."
+ exit 1
+fi
+
+# This finds 'card1' but ignores 'card1-DP-6'
+CARD_NAME=$(ls "/sys/bus/pci/devices/$PCI_ID/drm" | grep -E '^card[0-9]+$' | head -n 1)
+
+if [ -z "$CARD_NAME" ]; then
+ echo "Error: Could not determine card name for $PCI_ID"
+ exit 1
+fi
+
+# Construct the clean path (e.g., /sys/class/drm/card1)
+CARD_PATH="/sys/class/drm/$CARD_NAME"
+
+echo "Tuning GPU: $CARD_NAME (at $CARD_PATH)..."
+
+# 2. Force Manual Performance Level (Required for UV)
+# We use 'tee' without pipe to ensure exact errors are caught, but your 'echo | sudo tee' method is fine.
+echo "manual" | sudo tee "$CARD_PATH/device/power_dpm_force_performance_level" > /dev/null
+if [ $? -ne 0 ]; then echo "Failed to set Manual mode. Check permissions/path."; exit 1; fi
+
+# 3. Apply -75mV Undervolt
+echo "vo -75" | sudo tee "$CARD_PATH/device/pp_od_clk_voltage" > /dev/null
+echo "c" | sudo tee "$CARD_PATH/device/pp_od_clk_voltage" > /dev/null
+echo "Applied Undervolt (-75mV)"
+
+# 4. Set Power Limit to 315W
+# Find the hwmon directory strictly inside the device
+HWMON_DIR=$(find "$CARD_PATH/device/hwmon/" -maxdepth 1 -name "hwmon*" | head -n 1)
+if [ -n "$HWMON_DIR" ]; then
+ echo "315000000" | sudo tee "$HWMON_DIR/power1_cap" > /dev/null
+ echo "Applied Power Limit (315W)"
+else
+ echo "Error: Could not find hwmon directory for power limit."
+fi
+```
+
+### Usage
+
+```bash
+chmod +x tune_r9700.sh
+sudo ./tune_r9700.sh
+```
+
+## 4\. Verification
+
+Run these commands to confirm settings are active.
+
+**Check Undervolt:**
+
+```bash
+# Look for: OD_VDDGFX_OFFSET: -75mV
+cat /sys/class/drm/card0/device/pp_od_clk_voltage
+```
+
+**Check Power Limit:**
+
+```bash
+# Should be ~315 W
+cat /sys/class/drm/card0/device/hwmon/hwmon*/power1_cap
+```
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..7a531dc
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 14:59:00 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-ce1f54c0-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 200.72 \nTotal input tokens: 38358 \nTotal generated tokens: 40296 \nRequest throughput (req/s): 0.90 \nOutput token throughput (tok/s): 200.76 \nPeak output token throughput (tok/s): 344.00 \nPeak concurrent requests: 16.00 \nTotal token throughput (tok/s): 391.86 \n---------------Time to First Token----------------\nMean TTFT (ms): 104.14 \nMedian TTFT (ms): 87.12 \nP99 TTFT (ms): 220.96 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 41.41 \nMedian TPOT (ms): 41.23 \nP99 TPOT (ms): 46.01 \n---------------Inter-token Latency----------------\nMean ITL (ms): 41.43 \nMedian ITL (ms): 40.25 \nP99 ITL (ms): 86.38 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..2621b7e
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 15:02:31 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-959ca2a7-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 668 \nFailed requests: 52 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 180.01 \nTotal input tokens: 137901 \nTotal generated tokens: 128399 \nRequest throughput (req/s): 3.71 \nOutput token throughput (tok/s): 713.31 \nPeak output token throughput (tok/s): 1215.00 \nPeak concurrent requests: 72.00 \nTotal token throughput (tok/s): 1479.40 \n---------------Time to First Token----------------\nMean TTFT (ms): 123.92 \nMedian TTFT (ms): 95.36 \nP99 TTFT (ms): 403.64 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 73.89 \nMedian TPOT (ms): 55.38 \nP99 TPOT (ms): 442.43 \n---------------Inter-token Latency----------------\nMean ITL (ms): 54.97 \nMedian ITL (ms): 49.28 \nP99 ITL (ms): 187.29 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..f9a58ad
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,16815 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 14:58:16 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:16 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:16 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=58748)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:20 [model.py:514] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:20 [model.py:1636] Using max model len 32768
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:20 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:24 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:24 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:48181 backend=nccl
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:24 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:24 [gpu_model_runner.py:3562] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:25 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 3.07it/s]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.11it/s]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:03<00:01, 1.14s/it]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:04<00:00, 1.23s/it]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:04<00:00, 1.11s/it]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:30 [default_loader.py:308] Loading weights took 4.53 seconds
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:30 [gpu_model_runner.py:3659] Model loading took 15.4180 GiB memory and 5.228343 seconds
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:36 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/824e5832da/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:36 [backends.py:694] Dynamo bytecode transform time: 5.17 s
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:38 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:43 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 4.31 s
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:43 [monitor.py:34] torch.compile takes 9.49 s in total
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:45 [gpu_worker.py:375] Available KV cache memory: 14.82 GiB
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:45 [kv_cache_utils.py:1291] GPU KV cache size: 97,088 tokens
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:45 [kv_cache_utils.py:1296] Maximum concurrency for 32,768 tokens per request: 2.96x
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 17.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 19.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 20.75it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 20.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 20.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 21.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 20.75it/s]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 5.45it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 14.01it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:00<00:00, 17.78it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:00<00:00, 20.16it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 17.97it/s]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:48 [gpu_model_runner.py:4610] Graph capturing finished in 2 secs, took 1.04 GiB
+[0;36m(EngineCore_DP0 pid=58913)[0;0m INFO 12-19 14:58:48 [core.py:259] init engine (profile, create kv cache, warmup model) took 17.39 seconds
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:49 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=58748)[0;0m WARNING 12-19 14:58:49 [model.py:1462] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:49 [serving_responses.py:201] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:49 [serving_chat.py:137] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [serving_completion.py:77] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [serving_chat.py:137] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:58:50 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=58748)[0;0m INFO: Started server process [58748]
+[0;36m(APIServer pid=58748)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=58748)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:41284 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:59:10 [loggers.py:248] Engine 000: Avg prompt throughput: 7.4 tokens/s, Avg generation throughput: 22.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:59:20 [loggers.py:248] Engine 000: Avg prompt throughput: 131.2 tokens/s, Avg generation throughput: 114.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:46026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:46026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:59:30 [loggers.py:248] Engine 000: Avg prompt throughput: 224.6 tokens/s, Avg generation throughput: 163.7 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:59:40 [loggers.py:248] Engine 000: Avg prompt throughput: 279.4 tokens/s, Avg generation throughput: 209.9 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 14:59:50 [loggers.py:248] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 175.6 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:46026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:00:00 [loggers.py:248] Engine 000: Avg prompt throughput: 375.3 tokens/s, Avg generation throughput: 211.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:00:10 [loggers.py:248] Engine 000: Avg prompt throughput: 405.3 tokens/s, Avg generation throughput: 229.8 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:00:20 [loggers.py:248] Engine 000: Avg prompt throughput: 215.3 tokens/s, Avg generation throughput: 206.3 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:00:30 [loggers.py:248] Engine 000: Avg prompt throughput: 298.4 tokens/s, Avg generation throughput: 214.2 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:00:40 [loggers.py:248] Engine 000: Avg prompt throughput: 301.7 tokens/s, Avg generation throughput: 287.0 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:00:50 [loggers.py:248] Engine 000: Avg prompt throughput: 156.0 tokens/s, Avg generation throughput: 311.6 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:01:00 [loggers.py:248] Engine 000: Avg prompt throughput: 41.1 tokens/s, Avg generation throughput: 249.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:01:10 [loggers.py:248] Engine 000: Avg prompt throughput: 232.2 tokens/s, Avg generation throughput: 195.4 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:01:20 [loggers.py:248] Engine 000: Avg prompt throughput: 219.2 tokens/s, Avg generation throughput: 243.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:01:30 [loggers.py:248] Engine 000: Avg prompt throughput: 152.8 tokens/s, Avg generation throughput: 244.1 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:32768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:01:40 [loggers.py:248] Engine 000: Avg prompt throughput: 70.6 tokens/s, Avg generation throughput: 153.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:57840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:45828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:01:50 [loggers.py:248] Engine 000: Avg prompt throughput: 172.9 tokens/s, Avg generation throughput: 174.6 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:47358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:02:00 [loggers.py:248] Engine 000: Avg prompt throughput: 130.7 tokens/s, Avg generation throughput: 178.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:33208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:02:10 [loggers.py:248] Engine 000: Avg prompt throughput: 205.6 tokens/s, Avg generation throughput: 257.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:02:20 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 150.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:02:30 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 48.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:02:40 [loggers.py:248] Engine 000: Avg prompt throughput: 116.7 tokens/s, Avg generation throughput: 32.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 2.8%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:02:50 [loggers.py:248] Engine 000: Avg prompt throughput: 872.1 tokens/s, Avg generation throughput: 362.0 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.9%, Prefix cache hit rate: 19.7%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:03:00 [loggers.py:248] Engine 000: Avg prompt throughput: 1193.2 tokens/s, Avg generation throughput: 704.1 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.6%, Prefix cache hit rate: 34.9%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:03:10 [loggers.py:248] Engine 000: Avg prompt throughput: 863.9 tokens/s, Avg generation throughput: 801.8 tokens/s, Running: 46 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.4%, Prefix cache hit rate: 42.5%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:03:20 [loggers.py:248] Engine 000: Avg prompt throughput: 465.3 tokens/s, Avg generation throughput: 907.6 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.7%, Prefix cache hit rate: 45.9%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:03:30 [loggers.py:248] Engine 000: Avg prompt throughput: 902.6 tokens/s, Avg generation throughput: 804.9 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.0%, Prefix cache hit rate: 44.6%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:03:40 [loggers.py:248] Engine 000: Avg prompt throughput: 1066.5 tokens/s, Avg generation throughput: 823.5 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.8%, Prefix cache hit rate: 39.5%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:03:50 [loggers.py:248] Engine 000: Avg prompt throughput: 615.9 tokens/s, Avg generation throughput: 907.6 tokens/s, Running: 50 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.7%, Prefix cache hit rate: 37.1%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:04:00 [loggers.py:248] Engine 000: Avg prompt throughput: 820.7 tokens/s, Avg generation throughput: 871.6 tokens/s, Running: 62 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.3%, Prefix cache hit rate: 34.2%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:04:10 [loggers.py:248] Engine 000: Avg prompt throughput: 713.4 tokens/s, Avg generation throughput: 1065.9 tokens/s, Running: 61 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.8%, Prefix cache hit rate: 32.1%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:04:20 [loggers.py:248] Engine 000: Avg prompt throughput: 894.7 tokens/s, Avg generation throughput: 917.3 tokens/s, Running: 60 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.2%, Prefix cache hit rate: 29.8%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:04:30 [loggers.py:248] Engine 000: Avg prompt throughput: 916.1 tokens/s, Avg generation throughput: 792.1 tokens/s, Running: 38 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.6%, Prefix cache hit rate: 27.7%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:04:40 [loggers.py:248] Engine 000: Avg prompt throughput: 364.6 tokens/s, Avg generation throughput: 790.1 tokens/s, Running: 32 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.6%, Prefix cache hit rate: 27.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:04:50 [loggers.py:248] Engine 000: Avg prompt throughput: 837.6 tokens/s, Avg generation throughput: 751.2 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.6%, Prefix cache hit rate: 25.6%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:05:00 [loggers.py:248] Engine 000: Avg prompt throughput: 1054.8 tokens/s, Avg generation throughput: 785.9 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.6%, Prefix cache hit rate: 23.8%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:05:10 [loggers.py:248] Engine 000: Avg prompt throughput: 585.7 tokens/s, Avg generation throughput: 731.1 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.1%, Prefix cache hit rate: 23.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:43374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:34282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO 12-19 15:05:20 [loggers.py:248] Engine 000: Avg prompt throughput: 740.6 tokens/s, Avg generation throughput: 771.9 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.0%, Prefix cache hit rate: 22.0%
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:38294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:60788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:51330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:37570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [dump_input.py:72] Dumping input data for V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '/home/kyuz0/.cache/vllm/torch_compile_cache/824e5832da', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': '/home/kyuz0/.cache/vllm/torch_compile_cache/824e5832da/rank_0_0/backbone'},
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [dump_input.py:79] Dumping scheduler output for model execution: SchedulerOutput(scheduled_new_reqs=[NewRequestData(req_id=cmpl-bench-959ca2a7-668-0,prompt_token_ids_len=597,mm_features=[],sampling_params=SamplingParams(n=1, presence_penalty=0.0, frequency_penalty=0.0, repetition_penalty=1.0, temperature=0.0, top_p=1.0, top_k=0, min_p=0.0, seed=None, stop=[], stop_token_ids=[151643], bad_words=[], include_stop_str_in_output=False, ignore_eos=False, max_tokens=11, min_tokens=0, logprobs=None, prompt_logprobs=None, skip_special_tokens=True, spaces_between_special_tokens=True, truncate_prompt_tokens=None, structured_outputs=None, extra_args=None),block_ids=([2457, 311, 4955, 4403, 501, 1834, 3391, 1462, 679, 1386, 2343, 978, 2420, 2359, 2291, 1348, 611, 4569, 4357, 1443, 2301, 4246, 3135, 4341, 5307, 1176, 1744, 4249, 2188, 1792, 1334, 4950, 562, 5082, 127, 4989, 4837, 4530],),num_computed_tokens=0,lora_request=None,prompt_embeds_shape=None)], scheduled_cached_reqs=CachedRequestData(req_ids=['cmpl-bench-959ca2a7-518-0', 'cmpl-bench-959ca2a7-553-0', 'cmpl-bench-959ca2a7-573-0', 'cmpl-bench-959ca2a7-582-0', 'cmpl-bench-959ca2a7-586-0', 'cmpl-bench-959ca2a7-596-0', 'cmpl-bench-959ca2a7-602-0', 'cmpl-bench-959ca2a7-606-0', 'cmpl-bench-959ca2a7-607-0', 'cmpl-bench-959ca2a7-608-0', 'cmpl-bench-959ca2a7-609-0', 'cmpl-bench-959ca2a7-613-0', 'cmpl-bench-959ca2a7-617-0', 'cmpl-bench-959ca2a7-619-0', 'cmpl-bench-959ca2a7-622-0', 'cmpl-bench-959ca2a7-624-0', 'cmpl-bench-959ca2a7-625-0', 'cmpl-bench-959ca2a7-626-0', 'cmpl-bench-959ca2a7-627-0', 'cmpl-bench-959ca2a7-628-0', 'cmpl-bench-959ca2a7-630-0', 'cmpl-bench-959ca2a7-631-0', 'cmpl-bench-959ca2a7-632-0', 'cmpl-bench-959ca2a7-633-0', 'cmpl-bench-959ca2a7-634-0', 'cmpl-bench-959ca2a7-635-0', 'cmpl-bench-959ca2a7-636-0', 'cmpl-bench-959ca2a7-637-0', 'cmpl-bench-959ca2a7-638-0', 'cmpl-bench-959ca2a7-639-0', 'cmpl-bench-959ca2a7-640-0', 'cmpl-bench-959ca2a7-641-0', 'cmpl-bench-959ca2a7-642-0', 'cmpl-bench-959ca2a7-643-0', 'cmpl-bench-959ca2a7-648-0', 'cmpl-bench-959ca2a7-649-0', 'cmpl-bench-959ca2a7-650-0', 'cmpl-bench-959ca2a7-655-0', 'cmpl-bench-959ca2a7-656-0', 'cmpl-bench-959ca2a7-658-0', 'cmpl-bench-959ca2a7-659-0', 'cmpl-bench-959ca2a7-660-0', 'cmpl-bench-959ca2a7-661-0', 'cmpl-bench-959ca2a7-662-0', 'cmpl-bench-959ca2a7-663-0', 'cmpl-bench-959ca2a7-664-0', 'cmpl-bench-959ca2a7-665-0', 'cmpl-bench-959ca2a7-666-0', 'cmpl-bench-959ca2a7-667-0'], resumed_req_ids=[], new_token_ids=[], all_token_ids={}, new_block_ids=[null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, null, [[585]], null, null, null, null, null, [[1203]], null, null, null, null, null, [[4226]], null, null, null, null, [[5315]], null, null, null, null, null, null, null, null], num_computed_tokens=[743, 623, 1340, 663, 450, 395, 767, 317, 374, 461, 386, 385, 486, 541, 231, 211, 856, 191, 183, 185, 204, 191, 242, 224, 166, 168, 173, 157, 951, 352, 143, 121, 230, 115, 193, 80, 129, 58, 175, 52, 32, 104, 25, 25, 781, 533, 22, 622, 583], num_output_tokens=[738, 606, 483, 434, 426, 353, 318, 312, 311, 307, 305, 285, 277, 238, 227, 198, 186, 181, 175, 173, 172, 171, 171, 167, 159, 143, 142, 142, 137, 133, 124, 110, 89, 86, 65, 65, 61, 49, 48, 45, 25, 23, 12, 12, 10, 9, 8, 4, 3]), num_scheduled_tokens={cmpl-bench-959ca2a7-613-0: 1, cmpl-bench-959ca2a7-602-0: 1, cmpl-bench-959ca2a7-622-0: 1, cmpl-bench-959ca2a7-518-0: 1, cmpl-bench-959ca2a7-617-0: 1, cmpl-bench-959ca2a7-655-0: 1, cmpl-bench-959ca2a7-643-0: 1, cmpl-bench-959ca2a7-663-0: 1, cmpl-bench-959ca2a7-596-0: 1, cmpl-bench-959ca2a7-637-0: 1, cmpl-bench-959ca2a7-631-0: 1, cmpl-bench-959ca2a7-638-0: 1, cmpl-bench-959ca2a7-573-0: 1, cmpl-bench-959ca2a7-642-0: 1, cmpl-bench-959ca2a7-619-0: 1, cmpl-bench-959ca2a7-607-0: 1, cmpl-bench-959ca2a7-658-0: 1, cmpl-bench-959ca2a7-635-0: 1, cmpl-bench-959ca2a7-632-0: 1, cmpl-bench-959ca2a7-634-0: 1, cmpl-bench-959ca2a7-633-0: 1, cmpl-bench-959ca2a7-662-0: 1, cmpl-bench-959ca2a7-659-0: 1, cmpl-bench-959ca2a7-630-0: 1, cmpl-bench-959ca2a7-624-0: 1, cmpl-bench-959ca2a7-628-0: 1, cmpl-bench-959ca2a7-667-0: 1, cmpl-bench-959ca2a7-606-0: 1, cmpl-bench-959ca2a7-626-0: 1, cmpl-bench-959ca2a7-553-0: 1, cmpl-bench-959ca2a7-608-0: 1, cmpl-bench-959ca2a7-661-0: 1, cmpl-bench-959ca2a7-664-0: 1, cmpl-bench-959ca2a7-627-0: 1, cmpl-bench-959ca2a7-641-0: 1, cmpl-bench-959ca2a7-649-0: 1, cmpl-bench-959ca2a7-656-0: 1, cmpl-bench-959ca2a7-636-0: 1, cmpl-bench-959ca2a7-625-0: 1, cmpl-bench-959ca2a7-639-0: 1, cmpl-bench-959ca2a7-648-0: 1, cmpl-bench-959ca2a7-666-0: 1, cmpl-bench-959ca2a7-609-0: 1, cmpl-bench-959ca2a7-665-0: 1, cmpl-bench-959ca2a7-668-0: 597, cmpl-bench-959ca2a7-660-0: 1, cmpl-bench-959ca2a7-640-0: 1, cmpl-bench-959ca2a7-586-0: 1, cmpl-bench-959ca2a7-650-0: 1, cmpl-bench-959ca2a7-582-0: 1}, total_num_scheduled_tokens=646, scheduled_spec_decode_tokens={}, scheduled_encoder_inputs={}, num_common_prefix_blocks=[0], finished_req_ids=[], free_encoder_mm_hashes=[], preempted_req_ids=[], pending_structured_output_tokens=false, kv_connector_metadata=null, ec_connector_metadata=null)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [dump_input.py:81] Dumping scheduler stats: SchedulerStats(num_running_reqs=50, num_waiting_reqs=0, step_counter=0, current_wave=0, kv_cache_usage=0.181803197626504, prefix_cache_stats=PrefixCacheStats(reset=False, requests=1, queries=597, hits=0, preempted_requests=0, preempted_queries=0, preempted_hits=0), connector_prefix_cache_stats=None, kv_cache_eviction_events=[], spec_decoding_stats=None, kv_connector_stats=None, waiting_lora_adapters={}, running_lora_adapters={}, cudagraph_stats=None)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] EngineCore encountered a fatal error.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 859, in run_engine_core
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] engine_core.run_busy_loop()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 886, in run_busy_loop
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] self._process_engine_step()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 919, in _process_engine_step
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] outputs, model_executed = self.step_fn()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 353, in step
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] model_output = self.model_executor.sample_tokens(grammar_output)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/uniproc_executor.py", line 110, in sample_tokens
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] return self.collective_rpc(
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] "sample_tokens",
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] single_value=True,
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] )
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/uniproc_executor.py", line 75, in collective_rpc
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] result = run_method(self.driver_worker, method, args, kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/serial_utils.py", line 461, in run_method
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] return func(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] return func(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 572, in sample_tokens
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] return self.model_runner.sample_tokens(grammar_output)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] return func(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 3281, in sample_tokens
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ) = self._bookkeeping_sync(
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] scheduler_output,
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ...<4 lines>...
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] spec_decode_metadata,
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] )
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 2625, in _bookkeeping_sync
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] valid_sampled_token_ids = self._to_list(sampled_token_ids)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 5492, in _to_list
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] self.transfer_event.synchronize()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] torch.AcceleratorError: HIP error: an illegal memory access was encountered
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] Search for `hipErrorIllegalAddress' in https://rocm.docs.amd.com/projects/HIP/en/latest/index.html for more information.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] HIP kernel errors might be asynchronously reported at some other API call, so the stacktrace below might be incorrect.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] For debugging consider passing AMD_SERIALIZE_KERNEL=3
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868] Compile with `TORCH_USE_HIP_DSA` to enable device-side assertions.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ERROR 12-19 15:05:29 [core.py:868]
+[0;36m(EngineCore_DP0 pid=58913)[0;0m Process EngineCore_DP0:
+[0;36m(EngineCore_DP0 pid=58913)[0;0m Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 313, in _bootstrap
+[0;36m(EngineCore_DP0 pid=58913)[0;0m self.run()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 108, in run
+[0;36m(EngineCore_DP0 pid=58913)[0;0m self._target(*self._args, **self._kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 870, in run_engine_core
+[0;36m(EngineCore_DP0 pid=58913)[0;0m raise e
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 859, in run_engine_core
+[0;36m(EngineCore_DP0 pid=58913)[0;0m engine_core.run_busy_loop()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 886, in run_busy_loop
+[0;36m(EngineCore_DP0 pid=58913)[0;0m self._process_engine_step()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 919, in _process_engine_step
+[0;36m(EngineCore_DP0 pid=58913)[0;0m outputs, model_executed = self.step_fn()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 353, in step
+[0;36m(EngineCore_DP0 pid=58913)[0;0m model_output = self.model_executor.sample_tokens(grammar_output)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/uniproc_executor.py", line 110, in sample_tokens
+[0;36m(EngineCore_DP0 pid=58913)[0;0m return self.collective_rpc(
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m "sample_tokens",
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=58913)[0;0m single_value=True,
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m )
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/uniproc_executor.py", line 75, in collective_rpc
+[0;36m(EngineCore_DP0 pid=58913)[0;0m result = run_method(self.driver_worker, method, args, kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/serial_utils.py", line 461, in run_method
+[0;36m(EngineCore_DP0 pid=58913)[0;0m return func(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(EngineCore_DP0 pid=58913)[0;0m return func(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 572, in sample_tokens
+[0;36m(EngineCore_DP0 pid=58913)[0;0m return self.model_runner.sample_tokens(grammar_output)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(EngineCore_DP0 pid=58913)[0;0m return func(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 3281, in sample_tokens
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ) = self._bookkeeping_sync(
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m scheduler_output,
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ...<4 lines>...
+[0;36m(EngineCore_DP0 pid=58913)[0;0m spec_decode_metadata,
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m )
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 2625, in _bookkeeping_sync
+[0;36m(EngineCore_DP0 pid=58913)[0;0m valid_sampled_token_ids = self._to_list(sampled_token_ids)
+[0;36m(EngineCore_DP0 pid=58913)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 5492, in _to_list
+[0;36m(EngineCore_DP0 pid=58913)[0;0m self.transfer_event.synchronize()
+[0;36m(EngineCore_DP0 pid=58913)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=58913)[0;0m torch.AcceleratorError: HIP error: an illegal memory access was encountered
+[0;36m(EngineCore_DP0 pid=58913)[0;0m Search for `hipErrorIllegalAddress' in https://rocm.docs.amd.com/projects/HIP/en/latest/index.html for more information.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m HIP kernel errors might be asynchronously reported at some other API call, so the stacktrace below might be incorrect.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m For debugging consider passing AMD_SERIALIZE_KERNEL=3
+[0;36m(EngineCore_DP0 pid=58913)[0;0m Compile with `TORCH_USE_HIP_DSA` to enable device-side assertions.
+[0;36m(EngineCore_DP0 pid=58913)[0;0m
+[rank0]:[W1219 15:05:29.875861010 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] AsyncLLM output_handler failed.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [async_llm.py:538] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+hipModuleUnload failed:
+ error: an illegal memory access was encountered
+hipModuleUnload failed:
+ error: an illegal memory access was encountered
+hipModuleUnload failed:
+ error: an illegal memory access was encountered
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:29 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=58748)[0;0m ERROR 12-19 15:05:30 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:50762 - "POST /v1/completions HTTP/1.1" 500 Internal Server Error
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52514 - "POST /v1/completions HTTP/1.1" 500 Internal Server Error
+[0;36m(APIServer pid=58748)[0;0m INFO: 127.0.0.1:52528 - "POST /v1/completions HTTP/1.1" 500 Internal Server Error
+[0;36m(APIServer pid=58748)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=58748)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=58748)[0;0m INFO: Application shutdown complete.
+[0;36m(APIServer pid=58748)[0;0m INFO: Finished server process [58748]
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..e791f92
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 518.4320176600013,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 1.9288932124863885,
+ "tokens_per_second": 1429.9541207853842
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_server.log
new file mode 100644
index 0000000..d024178
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_server.log
@@ -0,0 +1,432 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 17:09:03 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=75372)[0;0m INFO 12-19 17:09:03 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=75372)[0;0m INFO 12-19 17:09:03 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32000, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=75372)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=75372)[0;0m INFO 12-19 17:09:08 [model.py:514] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=75372)[0;0m INFO 12-19 17:09:08 [model.py:1636] Using max model len 32000
+[0;36m(APIServer pid=75372)[0;0m INFO 12-19 17:09:08 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=75534)[0;0m INFO 12-19 17:09:12 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32000, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=75534)[0;0m WARNING 12-19 17:09:12 [multiproc_executor.py:884] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+INFO 12-19 17:09:15 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:35477 backend=nccl
+INFO 12-19 17:09:15 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:35477 backend=nccl
+INFO 12-19 17:09:16 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-19 17:09:16 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-19 17:09:16 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:17 [gpu_model_runner.py:3562] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(Worker_TP1 pid=75617)[0;0m INFO 12-19 17:09:17 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:17 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=75616)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=75616)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:01, 2.57it/s]
+[0;36m(Worker_TP0 pid=75616)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.02it/s]
+[0;36m(Worker_TP0 pid=75616)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:03<00:01, 1.14s/it]
+[0;36m(Worker_TP0 pid=75616)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:04<00:00, 1.19s/it]
+[0;36m(Worker_TP0 pid=75616)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:04<00:00, 1.09s/it]
+[0;36m(Worker_TP0 pid=75616)[0;0m
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:22 [default_loader.py:308] Loading weights took 4.43 seconds
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:23 [gpu_model_runner.py:3659] Model loading took 7.8555 GiB memory and 5.221202 seconds
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:28 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/8760cb82b9/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:28 [backends.py:694] Dynamo bytecode transform time: 5.45 s
+[0;36m(Worker_TP1 pid=75617)[0;0m INFO 12-19 17:09:31 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:31 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:36 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 4.77 s
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:36 [monitor.py:34] torch.compile takes 10.22 s in total
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:39 [gpu_worker.py:375] Available KV cache memory: 22.49 GiB
+[0;36m(EngineCore_DP0 pid=75534)[0;0m INFO 12-19 17:09:39 [kv_cache_utils.py:1291] GPU KV cache size: 294,752 tokens
+[0;36m(EngineCore_DP0 pid=75534)[0;0m INFO 12-19 17:09:39 [kv_cache_utils.py:1296] Maximum concurrency for 32,000 tokens per request: 9.21x
+[0;36m(Worker_TP0 pid=75616)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:08, 2.07it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:07, 2.15it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:01<00:07, 2.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:01<00:06, 2.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:02<00:05, 2.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:02<00:05, 2.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:02<00:04, 2.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:03<00:04, 2.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:03<00:04, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:04<00:03, 2.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:04<00:03, 2.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:04<00:02, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:05<00:02, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:05<00:02, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:06<00:01, 2.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:06<00:01, 2.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:06<00:00, 2.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:07<00:00, 2.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.51it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.44it/s]
+[0;36m(Worker_TP0 pid=75616)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:06, 1.63it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:01<00:04, 1.99it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:01<00:03, 2.20it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:01<00:03, 2.30it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:02<00:02, 2.36it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:02<00:02, 2.40it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:03<00:01, 2.44it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:03<00:01, 2.47it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:03<00:00, 2.50it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:04<00:00, 2.51it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.53it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.39it/s]
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:52 [gpu_model_runner.py:4610] Graph capturing finished in 13 secs, took 0.66 GiB
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] WorkerProc hit an exception.
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 76.00 MiB. GPU 1 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.75 GiB is allocated by PyTorch, with 68.00 MiB allocated in private pools (e.g., HIP Graphs), and 167.43 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 76.00 MiB. GPU 1 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.75 GiB is allocated by PyTorch, with 68.00 MiB allocated in private pools (e.g., HIP Graphs), and 167.43 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP1 pid=75617)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] WorkerProc hit an exception.
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 38.00 MiB. GPU 0 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.83 GiB is allocated by PyTorch, with 68.00 MiB allocated in private pools (e.g., HIP Graphs), and 169.24 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 38.00 MiB. GPU 0 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.83 GiB is allocated by PyTorch, with 68.00 MiB allocated in private pools (e.g., HIP Graphs), and 169.24 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP0 pid=75616)[0;0m ERROR 12-19 17:09:52 [multiproc_executor.py:826]
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] EngineCore failed to start.
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 857, in run_engine_core
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] engine_core = EngineCoreProc(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 637, in __init__
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] super().__init__(
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] vllm_config, executor_class, log_stats, executor_fail_callback
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 109, in __init__
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] num_gpu_blocks, num_cpu_blocks, kv_cache_config = self._initialize_kv_caches(
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] vllm_config
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 256, in _initialize_kv_caches
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] self.model_executor.initialize_from_config(kv_cache_configs)
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/abstract.py", line 116, in initialize_from_config
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] self.collective_rpc("compile_or_warm_up_model")
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 344, in get_response
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] raise RuntimeError(
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ERROR 12-19 17:09:52 [core.py:866] RuntimeError: Worker failed with error 'CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.', please check the stack trace above for the root cause
+[0;36m(EngineCore_DP0 pid=75534)[0;0m Process EngineCore_DP0:
+[0;36m(EngineCore_DP0 pid=75534)[0;0m Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 313, in _bootstrap
+[0;36m(EngineCore_DP0 pid=75534)[0;0m self.run()
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 108, in run
+[0;36m(EngineCore_DP0 pid=75534)[0;0m self._target(*self._args, **self._kwargs)
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 870, in run_engine_core
+[0;36m(EngineCore_DP0 pid=75534)[0;0m raise e
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 857, in run_engine_core
+[0;36m(EngineCore_DP0 pid=75534)[0;0m engine_core = EngineCoreProc(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 637, in __init__
+[0;36m(EngineCore_DP0 pid=75534)[0;0m super().__init__(
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m vllm_config, executor_class, log_stats, executor_fail_callback
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m )
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 109, in __init__
+[0;36m(EngineCore_DP0 pid=75534)[0;0m num_gpu_blocks, num_cpu_blocks, kv_cache_config = self._initialize_kv_caches(
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m vllm_config
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m )
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 256, in _initialize_kv_caches
+[0;36m(EngineCore_DP0 pid=75534)[0;0m self.model_executor.initialize_from_config(kv_cache_configs)
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/abstract.py", line 116, in initialize_from_config
+[0;36m(EngineCore_DP0 pid=75534)[0;0m self.collective_rpc("compile_or_warm_up_model")
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=75534)[0;0m return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=75534)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 344, in get_response
+[0;36m(EngineCore_DP0 pid=75534)[0;0m raise RuntimeError(
+[0;36m(EngineCore_DP0 pid=75534)[0;0m ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=75534)[0;0m )
+[0;36m(EngineCore_DP0 pid=75534)[0;0m RuntimeError: Worker failed with error 'CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.', please check the stack trace above for the root cause
+[0;36m(Worker_TP1 pid=75617)[0;0m INFO 12-19 17:09:52 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(Worker_TP0 pid=75616)[0;0m INFO 12-19 17:09:52 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(APIServer pid=75372)[0;0m Traceback (most recent call last):
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/bin/vllm", line 7, in
+[0;36m(APIServer pid=75372)[0;0m sys.exit(main())
+[0;36m(APIServer pid=75372)[0;0m ~~~~^^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/cli/main.py", line 73, in main
+[0;36m(APIServer pid=75372)[0;0m args.dispatch_function(args)
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/cli/serve.py", line 60, in cmd
+[0;36m(APIServer pid=75372)[0;0m uvloop.run(run_server(args))
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/uvloop/__init__.py", line 96, in run
+[0;36m(APIServer pid=75372)[0;0m return __asyncio.run(
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~^
+[0;36m(APIServer pid=75372)[0;0m wrapper(),
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m ...<2 lines>...
+[0;36m(APIServer pid=75372)[0;0m **run_kwargs
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m )
+[0;36m(APIServer pid=75372)[0;0m ^
+[0;36m(APIServer pid=75372)[0;0m File "/usr/lib64/python3.13/asyncio/runners.py", line 195, in run
+[0;36m(APIServer pid=75372)[0;0m return runner.run(main)
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/usr/lib64/python3.13/asyncio/runners.py", line 118, in run
+[0;36m(APIServer pid=75372)[0;0m return self._loop.run_until_complete(task)
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/uvloop/__init__.py", line 48, in wrapper
+[0;36m(APIServer pid=75372)[0;0m return await main
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 1398, in run_server
+[0;36m(APIServer pid=75372)[0;0m await run_server_worker(listen_address, sock, args, **uvicorn_kwargs)
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 1417, in run_server_worker
+[0;36m(APIServer pid=75372)[0;0m async with build_async_engine_client(
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=75372)[0;0m args,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^
+[0;36m(APIServer pid=75372)[0;0m client_config=client_config,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m ) as engine_client:
+[0;36m(APIServer pid=75372)[0;0m ^
+[0;36m(APIServer pid=75372)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 214, in __aenter__
+[0;36m(APIServer pid=75372)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 172, in build_async_engine_client
+[0;36m(APIServer pid=75372)[0;0m async with build_async_engine_client_from_engine_args(
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=75372)[0;0m engine_args,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m ...<2 lines>...
+[0;36m(APIServer pid=75372)[0;0m client_config=client_config,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m ) as engine:
+[0;36m(APIServer pid=75372)[0;0m ^
+[0;36m(APIServer pid=75372)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 214, in __aenter__
+[0;36m(APIServer pid=75372)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 213, in build_async_engine_client_from_engine_args
+[0;36m(APIServer pid=75372)[0;0m async_llm = AsyncLLM.from_vllm_config(
+[0;36m(APIServer pid=75372)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=75372)[0;0m ...<6 lines>...
+[0;36m(APIServer pid=75372)[0;0m client_index=client_index,
+[0;36m(APIServer pid=75372)[0;0m )
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 215, in from_vllm_config
+[0;36m(APIServer pid=75372)[0;0m return cls(
+[0;36m(APIServer pid=75372)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=75372)[0;0m ...<9 lines>...
+[0;36m(APIServer pid=75372)[0;0m client_index=client_index,
+[0;36m(APIServer pid=75372)[0;0m )
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 134, in __init__
+[0;36m(APIServer pid=75372)[0;0m self.engine_core = EngineCoreClient.make_async_mp_client(
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=75372)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m ...<4 lines>...
+[0;36m(APIServer pid=75372)[0;0m client_index=client_index,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m )
+[0;36m(APIServer pid=75372)[0;0m ^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 121, in make_async_mp_client
+[0;36m(APIServer pid=75372)[0;0m return AsyncMPClient(*client_args)
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 820, in __init__
+[0;36m(APIServer pid=75372)[0;0m super().__init__(
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=75372)[0;0m asyncio_mode=True,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m ...<3 lines>...
+[0;36m(APIServer pid=75372)[0;0m client_addresses=client_addresses,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m )
+[0;36m(APIServer pid=75372)[0;0m ^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 477, in __init__
+[0;36m(APIServer pid=75372)[0;0m with launch_core_engines(vllm_config, executor_class, log_stats) as (
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 148, in __exit__
+[0;36m(APIServer pid=75372)[0;0m next(self.gen)
+[0;36m(APIServer pid=75372)[0;0m ~~~~^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/utils.py", line 903, in launch_core_engines
+[0;36m(APIServer pid=75372)[0;0m wait_for_engine_startup(
+[0;36m(APIServer pid=75372)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=75372)[0;0m handshake_socket,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m ...<5 lines>...
+[0;36m(APIServer pid=75372)[0;0m coordinator.proc if coordinator else None,
+[0;36m(APIServer pid=75372)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=75372)[0;0m )
+[0;36m(APIServer pid=75372)[0;0m ^
+[0;36m(APIServer pid=75372)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/utils.py", line 960, in wait_for_engine_startup
+[0;36m(APIServer pid=75372)[0;0m raise RuntimeError(
+[0;36m(APIServer pid=75372)[0;0m ...<3 lines>...
+[0;36m(APIServer pid=75372)[0;0m )
+[0;36m(APIServer pid=75372)[0;0m RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {}
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_throughput.json
new file mode 100644
index 0000000..59fbb6e
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_Qwen3-14B-FP8-dynamic_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 470.41839592500037,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 2.1257672077931065,
+ "tokens_per_second": 1575.9035072220947
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..aa1d86d
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 15:31:43 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-242c7681-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 202.10 \nTotal input tokens: 40429 \nTotal generated tokens: 38744 \nRequest throughput (req/s): 0.89 \nOutput token throughput (tok/s): 191.71 \nPeak output token throughput (tok/s): 354.00 \nPeak concurrent requests: 17.00 \nTotal token throughput (tok/s): 391.75 \n---------------Time to First Token----------------\nMean TTFT (ms): 127.38 \nMedian TTFT (ms): 107.28 \nP99 TTFT (ms): 286.14 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 45.15 \nMedian TPOT (ms): 45.01 \nP99 TPOT (ms): 56.92 \n---------------Inter-token Latency----------------\nMean ITL (ms): 44.71 \nMedian ITL (ms): 42.72 \nP99 ITL (ms): 114.80 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..66a918c
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 15:35:15 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-8a860b01-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 224.44 \nTotal input tokens: 158086 \nTotal generated tokens: 153768 \nRequest throughput (req/s): 3.21 \nOutput token throughput (tok/s): 685.11 \nPeak output token throughput (tok/s): 1088.00 \nPeak concurrent requests: 111.00 \nTotal token throughput (tok/s): 1389.47 \n---------------Time to First Token----------------\nMean TTFT (ms): 3686.81 \nMedian TTFT (ms): 3310.63 \nP99 TTFT (ms): 10615.97 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 75.05 \nMedian TPOT (ms): 75.32 \nP99 TPOT (ms): 105.16 \n---------------Inter-token Latency----------------\nMean ITL (ms): 74.21 \nMedian ITL (ms): 62.11 \nP99 ITL (ms): 246.46 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..963249d
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1031 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 15:30:50 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:30:50 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:30:50 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 9900, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=60902)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:30:54 [model.py:514] Resolved architecture: Gemma3ForConditionalGeneration
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:30:54 [model.py:1636] Using max model len 9900
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:30:55 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:30:59 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='RedHatAI/gemma-3-12b-it-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/gemma-3-12b-it-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=9900, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/gemma-3-12b-it-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:00 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:36665 backend=nccl
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:00 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=61066)[0;0m Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:06 [gpu_model_runner.py:3562] Starting to load model RedHatAI/gemma-3-12b-it-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=61066)[0;0m WARNING 12-19 15:31:06 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:06 [layer.py:537] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(EngineCore_DP0 pid=61066)[0;0m WARNING 12-19 15:31:06 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:06 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=61066)[0;0m WARNING 12-19 15:31:06 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:01<00:03, 1.64s/it]
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:03<00:01, 1.56s/it]
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:04<00:00, 1.38s/it]
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:04<00:00, 1.43s/it]
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:11 [default_loader.py:308] Loading weights took 4.38 seconds
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:12 [gpu_model_runner.py:3659] Model loading took 13.5273 GiB memory and 4.987882 seconds
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:12 [gpu_model_runner.py:4446] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:20 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/dc55347db2/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:20 [backends.py:694] Dynamo bytecode transform time: 6.42 s
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:23 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:27 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 4.14 s
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:27 [monitor.py:34] torch.compile takes 10.56 s in total
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:30 [gpu_worker.py:375] Available KV cache memory: 16.25 GiB
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:30 [kv_cache_utils.py:1291] GPU KV cache size: 44,368 tokens
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:30 [kv_cache_utils.py:1296] Maximum concurrency for 9,900 tokens per request: 10.51x
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 19.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 21.69it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 22.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 23.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 24.23it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 24.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 23.99it/s]
+[0;36m(EngineCore_DP0 pid=61066)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 22.58it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 25.35it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 26.86it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 26.70it/s]
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:32 [gpu_model_runner.py:4610] Graph capturing finished in 2 secs, took 1.39 GiB
+[0;36m(EngineCore_DP0 pid=61066)[0;0m INFO 12-19 15:31:32 [core.py:259] init engine (profile, create kv cache, warmup model) took 20.40 seconds
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:33 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=60902)[0;0m INFO: Started server process [60902]
+[0;36m(APIServer pid=60902)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=60902)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:56338 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:31:54 [loggers.py:248] Engine 000: Avg prompt throughput: 8.6 tokens/s, Avg generation throughput: 31.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:32:04 [loggers.py:248] Engine 000: Avg prompt throughput: 138.5 tokens/s, Avg generation throughput: 111.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:32:14 [loggers.py:248] Engine 000: Avg prompt throughput: 239.1 tokens/s, Avg generation throughput: 166.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:32:24 [loggers.py:248] Engine 000: Avg prompt throughput: 297.3 tokens/s, Avg generation throughput: 199.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:32:34 [loggers.py:248] Engine 000: Avg prompt throughput: 202.2 tokens/s, Avg generation throughput: 172.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:56524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:32:44 [loggers.py:248] Engine 000: Avg prompt throughput: 397.0 tokens/s, Avg generation throughput: 178.3 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:32:54 [loggers.py:248] Engine 000: Avg prompt throughput: 380.0 tokens/s, Avg generation throughput: 219.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:56524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:33:04 [loggers.py:248] Engine 000: Avg prompt throughput: 170.5 tokens/s, Avg generation throughput: 200.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:36788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:33:14 [loggers.py:248] Engine 000: Avg prompt throughput: 397.2 tokens/s, Avg generation throughput: 172.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:33:24 [loggers.py:248] Engine 000: Avg prompt throughput: 196.2 tokens/s, Avg generation throughput: 246.1 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:33:34 [loggers.py:248] Engine 000: Avg prompt throughput: 326.2 tokens/s, Avg generation throughput: 279.2 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:33:44 [loggers.py:248] Engine 000: Avg prompt throughput: 39.6 tokens/s, Avg generation throughput: 289.9 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:33:54 [loggers.py:248] Engine 000: Avg prompt throughput: 226.1 tokens/s, Avg generation throughput: 177.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:50616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:34:04 [loggers.py:248] Engine 000: Avg prompt throughput: 330.3 tokens/s, Avg generation throughput: 251.5 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:34:14 [loggers.py:248] Engine 000: Avg prompt throughput: 70.2 tokens/s, Avg generation throughput: 203.8 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:34:24 [loggers.py:248] Engine 000: Avg prompt throughput: 107.3 tokens/s, Avg generation throughput: 193.6 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:50616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:34:34 [loggers.py:248] Engine 000: Avg prompt throughput: 149.1 tokens/s, Avg generation throughput: 177.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:34:44 [loggers.py:248] Engine 000: Avg prompt throughput: 184.1 tokens/s, Avg generation throughput: 182.5 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:43680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:34:54 [loggers.py:248] Engine 000: Avg prompt throughput: 184.5 tokens/s, Avg generation throughput: 243.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:35:04 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 153.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:35:14 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 37.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:35:24 [loggers.py:248] Engine 000: Avg prompt throughput: 91.1 tokens/s, Avg generation throughput: 28.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:35:34 [loggers.py:248] Engine 000: Avg prompt throughput: 828.1 tokens/s, Avg generation throughput: 288.9 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:35:44 [loggers.py:248] Engine 000: Avg prompt throughput: 1358.9 tokens/s, Avg generation throughput: 528.4 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:35:54 [loggers.py:248] Engine 000: Avg prompt throughput: 866.0 tokens/s, Avg generation throughput: 668.3 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:36:04 [loggers.py:248] Engine 000: Avg prompt throughput: 547.7 tokens/s, Avg generation throughput: 779.5 tokens/s, Running: 55 reqs, Waiting: 0 reqs, GPU KV cache usage: 43.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:36:14 [loggers.py:248] Engine 000: Avg prompt throughput: 985.5 tokens/s, Avg generation throughput: 737.6 tokens/s, Running: 64 reqs, Waiting: 4 reqs, GPU KV cache usage: 56.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:36:24 [loggers.py:248] Engine 000: Avg prompt throughput: 1040.5 tokens/s, Avg generation throughput: 729.6 tokens/s, Running: 64 reqs, Waiting: 8 reqs, GPU KV cache usage: 54.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:36:34 [loggers.py:248] Engine 000: Avg prompt throughput: 512.9 tokens/s, Avg generation throughput: 832.0 tokens/s, Running: 64 reqs, Waiting: 17 reqs, GPU KV cache usage: 61.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:36:44 [loggers.py:248] Engine 000: Avg prompt throughput: 675.3 tokens/s, Avg generation throughput: 812.8 tokens/s, Running: 64 reqs, Waiting: 23 reqs, GPU KV cache usage: 53.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:36:54 [loggers.py:248] Engine 000: Avg prompt throughput: 476.5 tokens/s, Avg generation throughput: 876.8 tokens/s, Running: 64 reqs, Waiting: 40 reqs, GPU KV cache usage: 48.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:37:04 [loggers.py:248] Engine 000: Avg prompt throughput: 724.6 tokens/s, Avg generation throughput: 889.6 tokens/s, Running: 64 reqs, Waiting: 42 reqs, GPU KV cache usage: 47.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:37:14 [loggers.py:248] Engine 000: Avg prompt throughput: 1012.9 tokens/s, Avg generation throughput: 819.2 tokens/s, Running: 64 reqs, Waiting: 26 reqs, GPU KV cache usage: 46.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:37:24 [loggers.py:248] Engine 000: Avg prompt throughput: 1177.0 tokens/s, Avg generation throughput: 768.0 tokens/s, Running: 62 reqs, Waiting: 14 reqs, GPU KV cache usage: 48.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:37:34 [loggers.py:248] Engine 000: Avg prompt throughput: 767.0 tokens/s, Avg generation throughput: 844.8 tokens/s, Running: 64 reqs, Waiting: 20 reqs, GPU KV cache usage: 51.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:59818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:37:44 [loggers.py:248] Engine 000: Avg prompt throughput: 644.1 tokens/s, Avg generation throughput: 870.4 tokens/s, Running: 62 reqs, Waiting: 23 reqs, GPU KV cache usage: 45.4%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:37:54 [loggers.py:248] Engine 000: Avg prompt throughput: 1269.4 tokens/s, Avg generation throughput: 767.9 tokens/s, Running: 64 reqs, Waiting: 7 reqs, GPU KV cache usage: 46.8%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:38:04 [loggers.py:248] Engine 000: Avg prompt throughput: 874.8 tokens/s, Avg generation throughput: 822.6 tokens/s, Running: 64 reqs, Waiting: 3 reqs, GPU KV cache usage: 48.8%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:38:14 [loggers.py:248] Engine 000: Avg prompt throughput: 412.4 tokens/s, Avg generation throughput: 908.0 tokens/s, Running: 64 reqs, Waiting: 18 reqs, GPU KV cache usage: 52.9%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:49228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:42850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:58898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:55950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:35844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:60772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:45610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO: 127.0.0.1:57900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:38:24 [loggers.py:248] Engine 000: Avg prompt throughput: 1276.2 tokens/s, Avg generation throughput: 697.6 tokens/s, Running: 63 reqs, Waiting: 20 reqs, GPU KV cache usage: 53.2%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:38:34 [loggers.py:248] Engine 000: Avg prompt throughput: 268.4 tokens/s, Avg generation throughput: 874.8 tokens/s, Running: 47 reqs, Waiting: 0 reqs, GPU KV cache usage: 43.0%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:38:44 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 573.3 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.3%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:38:54 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 221.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:39:04 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 42.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=60902)[0;0m INFO 12-19 15:39:06 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1219 15:39:06.109507194 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..8bfe3b4
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 666.9801866829985,
+ "num_requests": 1000,
+ "total_num_tokens": 755432,
+ "requests_per_second": 1.499294911552266,
+ "tokens_per_second": 1132.6153536237514
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..9c49747
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 18:04:12 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-b978cded-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 196.92 \nTotal input tokens: 40429 \nTotal generated tokens: 38727 \nRequest throughput (req/s): 0.91 \nOutput token throughput (tok/s): 196.66 \nPeak output token throughput (tok/s): 393.00 \nPeak concurrent requests: 17.00 \nTotal token throughput (tok/s): 401.97 \n---------------Time to First Token----------------\nMean TTFT (ms): 232.41 \nMedian TTFT (ms): 152.06 \nP99 TTFT (ms): 582.73 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 39.99 \nMedian TPOT (ms): 38.66 \nP99 TPOT (ms): 75.17 \n---------------Inter-token Latency----------------\nMean ITL (ms): 38.94 \nMedian ITL (ms): 33.55 \nP99 ITL (ms): 310.76 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..06f2684
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 18:07:39 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-3d8d3c55-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 299.13 \nTotal input tokens: 158086 \nTotal generated tokens: 152394 \nRequest throughput (req/s): 2.41 \nOutput token throughput (tok/s): 509.46 \nPeak output token throughput (tok/s): 1088.00 \nPeak concurrent requests: 294.00 \nTotal token throughput (tok/s): 1037.94 \n---------------Time to First Token----------------\nMean TTFT (ms): 37309.58 \nMedian TTFT (ms): 42371.37 \nP99 TTFT (ms): 86881.21 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 113.18 \nMedian TPOT (ms): 113.38 \nP99 TPOT (ms): 240.19 \n---------------Inter-token Latency----------------\nMean ITL (ms): 109.51 \nMedian ITL (ms): 58.62 \nP99 ITL (ms): 472.97 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log
new file mode 100644
index 0000000..04ff57b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log
@@ -0,0 +1,1053 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 18:02:43 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:02:43 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:02:43 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 9900, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=80667)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:02:47 [model.py:514] Resolved architecture: Gemma3ForConditionalGeneration
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:02:47 [model.py:1636] Using max model len 9900
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:02:48 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=80832)[0;0m INFO 12-19 18:02:52 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='RedHatAI/gemma-3-12b-it-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/gemma-3-12b-it-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=9900, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/gemma-3-12b-it-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=80832)[0;0m WARNING 12-19 18:02:52 [multiproc_executor.py:884] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+INFO 12-19 18:02:57 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:37123 backend=nccl
+INFO 12-19 18:02:57 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:37123 backend=nccl
+INFO 12-19 18:02:57 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-19 18:02:57 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-19 18:02:57 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:03 [gpu_model_runner.py:3562] Starting to load model RedHatAI/gemma-3-12b-it-FP8-dynamic...
+[0;36m(Worker_TP0 pid=80914)[0;0m WARNING 12-19 18:03:04 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:04 [layer.py:537] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP0 pid=80914)[0;0m WARNING 12-19 18:03:04 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP1 pid=80915)[0;0m WARNING 12-19 18:03:04 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP1 pid=80915)[0;0m INFO 12-19 18:03:04 [layer.py:537] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP1 pid=80915)[0;0m WARNING 12-19 18:03:04 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:04 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=80914)[0;0m WARNING 12-19 18:03:04 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP1 pid=80915)[0;0m INFO 12-19 18:03:04 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=80915)[0;0m WARNING 12-19 18:03:04 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP0 pid=80914)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=80914)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:01<00:02, 1.48s/it]
+[0;36m(Worker_TP0 pid=80914)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:02<00:01, 1.43s/it]
+[0;36m(Worker_TP0 pid=80914)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:03<00:00, 1.24s/it]
+[0;36m(Worker_TP0 pid=80914)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:03<00:00, 1.29s/it]
+[0;36m(Worker_TP0 pid=80914)[0;0m
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:08 [default_loader.py:308] Loading weights took 3.93 seconds
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:09 [gpu_model_runner.py:3659] Model loading took 7.1152 GiB memory and 4.668085 seconds
+[0;36m(Worker_TP1 pid=80915)[0;0m INFO 12-19 18:03:09 [gpu_model_runner.py:4446] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:09 [gpu_model_runner.py:4446] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:18 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/28de62fa48/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:18 [backends.py:694] Dynamo bytecode transform time: 7.17 s
+[0;36m(Worker_TP1 pid=80915)[0;0m INFO 12-19 18:03:22 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:22 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:40 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 18.55 s
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:40 [monitor.py:34] torch.compile takes 25.72 s in total
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:43 [gpu_worker.py:375] Available KV cache memory: 22.66 GiB
+[0;36m(EngineCore_DP0 pid=80832)[0;0m INFO 12-19 18:03:44 [kv_cache_utils.py:1291] GPU KV cache size: 123,744 tokens
+[0;36m(EngineCore_DP0 pid=80832)[0;0m INFO 12-19 18:03:44 [kv_cache_utils.py:1296] Maximum concurrency for 9,900 tokens per request: 29.30x
+[0;36m(Worker_TP0 pid=80914)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:07, 2.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:06, 2.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:01<00:06, 2.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:01<00:06, 2.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:02<00:05, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:02<00:05, 2.51it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:02<00:04, 2.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:03<00:04, 2.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:03<00:03, 2.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:03<00:03, 2.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:04<00:03, 2.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:04<00:02, 2.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:05<00:02, 2.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:05<00:01, 2.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:05<00:01, 2.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:06<00:01, 2.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:06<00:00, 2.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:07<00:00, 2.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.53it/s]
+[0;36m(Worker_TP0 pid=80914)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:04, 2.47it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:03, 2.55it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:01<00:03, 2.58it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:01<00:02, 2.62it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:01<00:02, 2.63it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:02<00:01, 2.64it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:02<00:01, 2.65it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:03<00:01, 2.66it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:03<00:00, 2.66it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:03<00:00, 2.68it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.70it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.65it/s]
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:03:56 [gpu_model_runner.py:4610] Graph capturing finished in 13 secs, took 1.03 GiB
+[0;36m(EngineCore_DP0 pid=80832)[0;0m INFO 12-19 18:03:56 [core.py:259] init engine (profile, create kv cache, warmup model) took 47.45 seconds
+[0;36m(EngineCore_DP0 pid=80832)[0;0m Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:03 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:04 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=80667)[0;0m INFO: Started server process [80667]
+[0;36m(APIServer pid=80667)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=80667)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39036 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:24 [loggers.py:248] Engine 000: Avg prompt throughput: 45.0 tokens/s, Avg generation throughput: 66.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:34 [loggers.py:248] Engine 000: Avg prompt throughput: 178.1 tokens/s, Avg generation throughput: 121.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:44 [loggers.py:248] Engine 000: Avg prompt throughput: 163.1 tokens/s, Avg generation throughput: 212.2 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:04:54 [loggers.py:248] Engine 000: Avg prompt throughput: 298.8 tokens/s, Avg generation throughput: 154.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:05:04 [loggers.py:248] Engine 000: Avg prompt throughput: 230.9 tokens/s, Avg generation throughput: 179.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:05:14 [loggers.py:248] Engine 000: Avg prompt throughput: 366.8 tokens/s, Avg generation throughput: 182.5 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:05:24 [loggers.py:248] Engine 000: Avg prompt throughput: 399.0 tokens/s, Avg generation throughput: 212.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:05:34 [loggers.py:248] Engine 000: Avg prompt throughput: 153.9 tokens/s, Avg generation throughput: 185.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:05:44 [loggers.py:248] Engine 000: Avg prompt throughput: 417.8 tokens/s, Avg generation throughput: 177.8 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:05:54 [loggers.py:248] Engine 000: Avg prompt throughput: 207.9 tokens/s, Avg generation throughput: 250.0 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:06:04 [loggers.py:248] Engine 000: Avg prompt throughput: 292.5 tokens/s, Avg generation throughput: 308.0 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:06:14 [loggers.py:248] Engine 000: Avg prompt throughput: 45.8 tokens/s, Avg generation throughput: 269.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:06:24 [loggers.py:248] Engine 000: Avg prompt throughput: 305.6 tokens/s, Avg generation throughput: 186.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:46692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:06:34 [loggers.py:248] Engine 000: Avg prompt throughput: 248.2 tokens/s, Avg generation throughput: 235.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:48768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:06:44 [loggers.py:248] Engine 000: Avg prompt throughput: 102.5 tokens/s, Avg generation throughput: 216.8 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:06:54 [loggers.py:248] Engine 000: Avg prompt throughput: 145.2 tokens/s, Avg generation throughput: 203.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:37572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:07:04 [loggers.py:248] Engine 000: Avg prompt throughput: 90.8 tokens/s, Avg generation throughput: 195.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:33762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:07:14 [loggers.py:248] Engine 000: Avg prompt throughput: 233.8 tokens/s, Avg generation throughput: 180.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:34916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:60744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:07:24 [loggers.py:248] Engine 000: Avg prompt throughput: 118.3 tokens/s, Avg generation throughput: 241.7 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:07:34 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 100.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:07:44 [loggers.py:248] Engine 000: Avg prompt throughput: 1.1 tokens/s, Avg generation throughput: 16.3 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:07:54 [loggers.py:248] Engine 000: Avg prompt throughput: 812.3 tokens/s, Avg generation throughput: 293.6 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.9%, Prefix cache hit rate: 16.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:08:04 [loggers.py:248] Engine 000: Avg prompt throughput: 1276.7 tokens/s, Avg generation throughput: 374.1 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.2%, Prefix cache hit rate: 32.9%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:08:14 [loggers.py:248] Engine 000: Avg prompt throughput: 712.2 tokens/s, Avg generation throughput: 449.9 tokens/s, Running: 54 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.3%, Prefix cache hit rate: 39.4%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:08:24 [loggers.py:248] Engine 000: Avg prompt throughput: 803.4 tokens/s, Avg generation throughput: 555.1 tokens/s, Running: 62 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.3%, Prefix cache hit rate: 45.4%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:08:34 [loggers.py:248] Engine 000: Avg prompt throughput: 438.3 tokens/s, Avg generation throughput: 684.7 tokens/s, Running: 64 reqs, Waiting: 23 reqs, GPU KV cache usage: 18.0%, Prefix cache hit rate: 48.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:08:44 [loggers.py:248] Engine 000: Avg prompt throughput: 549.9 tokens/s, Avg generation throughput: 543.9 tokens/s, Running: 64 reqs, Waiting: 44 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 45.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:08:54 [loggers.py:248] Engine 000: Avg prompt throughput: 483.4 tokens/s, Avg generation throughput: 524.7 tokens/s, Running: 63 reqs, Waiting: 52 reqs, GPU KV cache usage: 21.4%, Prefix cache hit rate: 42.6%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:09:04 [loggers.py:248] Engine 000: Avg prompt throughput: 663.4 tokens/s, Avg generation throughput: 486.4 tokens/s, Running: 63 reqs, Waiting: 70 reqs, GPU KV cache usage: 20.6%, Prefix cache hit rate: 39.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:09:14 [loggers.py:248] Engine 000: Avg prompt throughput: 488.2 tokens/s, Avg generation throughput: 556.8 tokens/s, Running: 63 reqs, Waiting: 100 reqs, GPU KV cache usage: 21.3%, Prefix cache hit rate: 37.8%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:09:24 [loggers.py:248] Engine 000: Avg prompt throughput: 410.3 tokens/s, Avg generation throughput: 498.4 tokens/s, Running: 63 reqs, Waiting: 110 reqs, GPU KV cache usage: 19.4%, Prefix cache hit rate: 36.4%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:09:34 [loggers.py:248] Engine 000: Avg prompt throughput: 313.8 tokens/s, Avg generation throughput: 608.0 tokens/s, Running: 63 reqs, Waiting: 133 reqs, GPU KV cache usage: 19.2%, Prefix cache hit rate: 35.3%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:09:44 [loggers.py:248] Engine 000: Avg prompt throughput: 417.5 tokens/s, Avg generation throughput: 652.8 tokens/s, Running: 63 reqs, Waiting: 142 reqs, GPU KV cache usage: 18.3%, Prefix cache hit rate: 34.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:09:54 [loggers.py:248] Engine 000: Avg prompt throughput: 526.9 tokens/s, Avg generation throughput: 544.0 tokens/s, Running: 64 reqs, Waiting: 154 reqs, GPU KV cache usage: 16.8%, Prefix cache hit rate: 32.5%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:59342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:10:04 [loggers.py:248] Engine 000: Avg prompt throughput: 420.8 tokens/s, Avg generation throughput: 544.0 tokens/s, Running: 63 reqs, Waiting: 176 reqs, GPU KV cache usage: 17.5%, Prefix cache hit rate: 31.4%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:10:14 [loggers.py:248] Engine 000: Avg prompt throughput: 560.4 tokens/s, Avg generation throughput: 518.4 tokens/s, Running: 64 reqs, Waiting: 180 reqs, GPU KV cache usage: 16.6%, Prefix cache hit rate: 30.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:39146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:54958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:10:24 [loggers.py:248] Engine 000: Avg prompt throughput: 661.7 tokens/s, Avg generation throughput: 486.4 tokens/s, Running: 62 reqs, Waiting: 185 reqs, GPU KV cache usage: 16.8%, Prefix cache hit rate: 28.6%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:51310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:10:34 [loggers.py:248] Engine 000: Avg prompt throughput: 879.9 tokens/s, Avg generation throughput: 403.2 tokens/s, Running: 64 reqs, Waiting: 193 reqs, GPU KV cache usage: 17.8%, Prefix cache hit rate: 26.9%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:55152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:53858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:41048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:50784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:56990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:44678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:58304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:40994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:57144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO: 127.0.0.1:42224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:10:44 [loggers.py:248] Engine 000: Avg prompt throughput: 332.6 tokens/s, Avg generation throughput: 569.6 tokens/s, Running: 62 reqs, Waiting: 227 reqs, GPU KV cache usage: 17.1%, Prefix cache hit rate: 26.3%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:10:54 [loggers.py:248] Engine 000: Avg prompt throughput: 592.1 tokens/s, Avg generation throughput: 550.4 tokens/s, Running: 63 reqs, Waiting: 202 reqs, GPU KV cache usage: 17.0%, Prefix cache hit rate: 25.3%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:11:04 [loggers.py:248] Engine 000: Avg prompt throughput: 434.5 tokens/s, Avg generation throughput: 550.4 tokens/s, Running: 63 reqs, Waiting: 176 reqs, GPU KV cache usage: 17.2%, Prefix cache hit rate: 24.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:11:14 [loggers.py:248] Engine 000: Avg prompt throughput: 805.4 tokens/s, Avg generation throughput: 390.4 tokens/s, Running: 64 reqs, Waiting: 146 reqs, GPU KV cache usage: 16.1%, Prefix cache hit rate: 23.5%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:11:24 [loggers.py:248] Engine 000: Avg prompt throughput: 560.6 tokens/s, Avg generation throughput: 505.6 tokens/s, Running: 63 reqs, Waiting: 121 reqs, GPU KV cache usage: 16.1%, Prefix cache hit rate: 22.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:11:34 [loggers.py:248] Engine 000: Avg prompt throughput: 606.0 tokens/s, Avg generation throughput: 563.1 tokens/s, Running: 63 reqs, Waiting: 94 reqs, GPU KV cache usage: 17.3%, Prefix cache hit rate: 22.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:11:44 [loggers.py:248] Engine 000: Avg prompt throughput: 400.2 tokens/s, Avg generation throughput: 627.1 tokens/s, Running: 64 reqs, Waiting: 72 reqs, GPU KV cache usage: 16.6%, Prefix cache hit rate: 21.5%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:11:54 [loggers.py:248] Engine 000: Avg prompt throughput: 406.9 tokens/s, Avg generation throughput: 640.0 tokens/s, Running: 62 reqs, Waiting: 52 reqs, GPU KV cache usage: 17.9%, Prefix cache hit rate: 21.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:12:04 [loggers.py:248] Engine 000: Avg prompt throughput: 794.4 tokens/s, Avg generation throughput: 416.0 tokens/s, Running: 63 reqs, Waiting: 25 reqs, GPU KV cache usage: 19.4%, Prefix cache hit rate: 20.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:12:14 [loggers.py:248] Engine 000: Avg prompt throughput: 455.7 tokens/s, Avg generation throughput: 583.6 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.0%, Prefix cache hit rate: 19.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:12:24 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 775.6 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.1%, Prefix cache hit rate: 19.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:12:34 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 298.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 19.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=80667)[0;0m INFO 12-19 18:12:43 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP1 pid=80915)[0;0m INFO 12-19 18:12:43 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(Worker_TP0 pid=80914)[0;0m INFO 12-19 18:12:43 [multiproc_executor.py:711] Parent process exited, terminating worker
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json
new file mode 100644
index 0000000..9610cd4
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 581.3710580350016,
+ "num_requests": 1000,
+ "total_num_tokens": 755432,
+ "requests_per_second": 1.72007186491178,
+ "tokens_per_second": 1299.397329054036
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..1b0ba43
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 17:37:17 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-27b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-0a333e52-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 45 \nFailed requests: 135 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 180.00 \nTotal input tokens: 10566 \nTotal generated tokens: 7279 \nRequest throughput (req/s): 0.25 \nOutput token throughput (tok/s): 40.44 \nPeak output token throughput (tok/s): 234.00 \nPeak concurrent requests: 14.00 \nTotal token throughput (tok/s): 99.14 \n---------------Time to First Token----------------\nMean TTFT (ms): 235.33 \nMedian TTFT (ms): 121.50 \nP99 TTFT (ms): 734.11 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 55.13 \nMedian TPOT (ms): 54.64 \nP99 TPOT (ms): 75.71 \n---------------Inter-token Latency----------------\nMean ITL (ms): 54.59 \nMedian ITL (ms): 50.79 \nP99 ITL (ms): 355.12 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..c0c69e0
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": false,
+ "raw_output": "WARNING 12-19 17:40:29 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-27b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-660ac132-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log
new file mode 100644
index 0000000..b16c6e3
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log
@@ -0,0 +1,2570 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 17:35:32 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:35:32 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:35:32 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/gemma-3-27b-it-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/gemma-3-27b-it-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 28900, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.94, 'max_num_seqs': 32}
+[0;36m(APIServer pid=77686)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:35:36 [model.py:514] Resolved architecture: Gemma3ForConditionalGeneration
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:35:36 [model.py:1636] Using max model len 28900
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:35:36 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=77851)[0;0m INFO 12-19 17:35:41 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='RedHatAI/gemma-3-27b-it-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/gemma-3-27b-it-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=28900, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/gemma-3-27b-it-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 64, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=77851)[0;0m WARNING 12-19 17:35:41 [multiproc_executor.py:884] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+INFO 12-19 17:35:46 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:33879 backend=nccl
+INFO 12-19 17:35:46 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:33879 backend=nccl
+INFO 12-19 17:35:46 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-19 17:35:46 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-19 17:35:46 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:35:52 [gpu_model_runner.py:3562] Starting to load model RedHatAI/gemma-3-27b-it-FP8-dynamic...
+[0;36m(Worker_TP0 pid=77933)[0;0m WARNING 12-19 17:35:53 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:35:53 [layer.py:537] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP0 pid=77933)[0;0m WARNING 12-19 17:35:53 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP1 pid=77934)[0;0m WARNING 12-19 17:35:53 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP1 pid=77934)[0;0m INFO 12-19 17:35:53 [layer.py:537] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP1 pid=77934)[0;0m WARNING 12-19 17:35:53 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP1 pid=77934)[0;0m INFO 12-19 17:35:53 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:35:53 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=77934)[0;0m WARNING 12-19 17:35:53 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP0 pid=77933)[0;0m WARNING 12-19 17:35:53 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/6 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 17% Completed | 1/6 [00:01<00:05, 1.06s/it]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 2/6 [00:02<00:04, 1.09s/it]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 3/6 [00:03<00:03, 1.10s/it]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 4/6 [00:04<00:02, 1.05s/it]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 83% Completed | 5/6 [00:05<00:01, 1.04s/it]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 6/6 [00:06<00:00, 1.12s/it]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 6/6 [00:06<00:00, 1.09s/it]
+[0;36m(Worker_TP0 pid=77933)[0;0m
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:00 [default_loader.py:308] Loading weights took 6.62 seconds
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:01 [gpu_model_runner.py:3659] Model loading took 14.2812 GiB memory and 7.486143 seconds
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:01 [gpu_model_runner.py:4446] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP1 pid=77934)[0;0m INFO 12-19 17:36:01 [gpu_model_runner.py:4446] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:12 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/2b20f74ec3/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:12 [backends.py:694] Dynamo bytecode transform time: 8.90 s
+[0;36m(Worker_TP1 pid=77934)[0;0m INFO 12-19 17:36:17 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:17 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:45 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 29.01 s
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:45 [monitor.py:34] torch.compile takes 37.90 s in total
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:36:49 [gpu_worker.py:375] Available KV cache memory: 14.58 GiB
+[0;36m(EngineCore_DP0 pid=77851)[0;0m WARNING 12-19 17:36:50 [kv_cache_utils.py:1033] Add 8 padding layers, may waste at most 15.38% KV cache memory
+[0;36m(EngineCore_DP0 pid=77851)[0;0m INFO 12-19 17:36:50 [kv_cache_utils.py:1291] GPU KV cache size: 54,464 tokens
+[0;36m(EngineCore_DP0 pid=77851)[0;0m INFO 12-19 17:36:50 [kv_cache_utils.py:1296] Maximum concurrency for 28,900 tokens per request: 8.04x
+[0;36m(Worker_TP0 pid=77933)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 9%|▉ | 1/11 [00:00<00:05, 1.83it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 2/11 [00:01<00:04, 1.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 3/11 [00:01<00:04, 1.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▋ | 4/11 [00:02<00:03, 1.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▌ | 5/11 [00:02<00:03, 1.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▍ | 6/11 [00:03<00:02, 1.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▎ | 7/11 [00:03<00:02, 1.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 8/11 [00:04<00:01, 1.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 9/11 [00:04<00:01, 1.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 91%|█████████ | 10/11 [00:05<00:00, 1.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 1.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 1.94it/s]
+[0;36m(Worker_TP0 pid=77933)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 14%|█▍ | 1/7 [00:00<00:03, 1.77it/s]
Capturing CUDA graphs (decode, FULL): 29%|██▊ | 2/7 [00:01<00:02, 1.89it/s]
Capturing CUDA graphs (decode, FULL): 43%|████▎ | 3/7 [00:01<00:02, 1.93it/s]
Capturing CUDA graphs (decode, FULL): 57%|█████▋ | 4/7 [00:02<00:01, 1.96it/s]
Capturing CUDA graphs (decode, FULL): 71%|███████▏ | 5/7 [00:02<00:01, 1.97it/s]
Capturing CUDA graphs (decode, FULL): 86%|████████▌ | 6/7 [00:03<00:00, 1.98it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:03<00:00, 1.98it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:03<00:00, 1.95it/s]
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:37:00 [gpu_model_runner.py:4610] Graph capturing finished in 10 secs, took 1.58 GiB
+[0;36m(EngineCore_DP0 pid=77851)[0;0m INFO 12-19 17:37:00 [core.py:259] init engine (profile, create kv cache, warmup model) took 59.41 seconds
+[0;36m(EngineCore_DP0 pid=77851)[0;0m Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:07 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=77686)[0;0m INFO: Started server process [77686]
+[0;36m(APIServer pid=77686)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=77686)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:54918 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:28 [loggers.py:248] Engine 000: Avg prompt throughput: 4.9 tokens/s, Avg generation throughput: 17.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:38 [loggers.py:248] Engine 000: Avg prompt throughput: 140.6 tokens/s, Avg generation throughput: 98.1 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:48 [loggers.py:248] Engine 000: Avg prompt throughput: 196.2 tokens/s, Avg generation throughput: 125.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:41430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:37:58 [loggers.py:248] Engine 000: Avg prompt throughput: 341.8 tokens/s, Avg generation throughput: 153.9 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:38:08 [loggers.py:248] Engine 000: Avg prompt throughput: 202.2 tokens/s, Avg generation throughput: 194.0 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:34404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] WorkerProc hit an exception.
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 572, in sample_tokens
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] return self.model_runner.sample_tokens(grammar_output)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 3281, in sample_tokens
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ) = self._bookkeeping_sync(
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] scheduler_output,
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] spec_decode_metadata,
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 2625, in _bookkeeping_sync
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] valid_sampled_token_ids = self._to_list(sampled_token_ids)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 5492, in _to_list
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] self.transfer_event.synchronize()
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] torch.AcceleratorError: HIP error: an illegal memory access was encountered
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] Search for `hipErrorIllegalAddress' in https://rocm.docs.amd.com/projects/HIP/en/latest/index.html for more information.
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] HIP kernel errors might be asynchronously reported at some other API call, so the stacktrace below might be incorrect.
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] For debugging consider passing AMD_SERIALIZE_KERNEL=3
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] Compile with `TORCH_USE_HIP_DSA` to enable device-side assertions.
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 572, in sample_tokens
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] return self.model_runner.sample_tokens(grammar_output)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 3281, in sample_tokens
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ) = self._bookkeeping_sync(
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] scheduler_output,
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] spec_decode_metadata,
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 2625, in _bookkeeping_sync
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] valid_sampled_token_ids = self._to_list(sampled_token_ids)
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 5492, in _to_list
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] self.transfer_event.synchronize()
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] torch.AcceleratorError: HIP error: an illegal memory access was encountered
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] Search for `hipErrorIllegalAddress' in https://rocm.docs.amd.com/projects/HIP/en/latest/index.html for more information.
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] HIP kernel errors might be asynchronously reported at some other API call, so the stacktrace below might be incorrect.
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] For debugging consider passing AMD_SERIALIZE_KERNEL=3
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826] Compile with `TORCH_USE_HIP_DSA` to enable device-side assertions.
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=77934)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:826]
+[rank1]:[E1219 17:38:16.877431190 ProcessGroupNCCL.cpp:2093] [PG ID 2 PG GUID 3 Rank 1] Process group watchdog thread terminated with exception: HIP error: an illegal memory access was encountered
+Search for `hipErrorIllegalAddress' in https://rocm.docs.amd.com/projects/HIP/en/latest/index.html for more information.
+HIP kernel errors might be asynchronously reported at some other API call, so the stacktrace below might be incorrect.
+For debugging consider passing AMD_SERIALIZE_KERNEL=3
+Compile with `TORCH_USE_HIP_DSA` to enable device-side assertions.
+
+Exception raised from getDevice at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/hip/impl/HIPGuardImplMasqueradingAsCUDA.h:75 (most recent call first):
+frame #0: c10::Error::Error(c10::SourceLocation, std::__cxx11::basic_string, std::allocator >) + 0x9d (0x7f01713fafdd in /opt/venv/lib64/python3.13/site-packages/torch/lib/libc10.so)
+frame #1: + 0x12568 (0x7f0171486568 in /opt/venv/lib64/python3.13/site-packages/torch/lib/libc10_hip.so)
+frame #2: c10d::ProcessGroupNCCL::Watchdog::runLoop() + 0x922 (0x7f0174a7ccc2 in /opt/venv/lib64/python3.13/site-packages/torch/lib/libtorch_hip.so)
+frame #3: c10d::ProcessGroupNCCL::Watchdog::run() + 0x105 (0x7f0174a7f1b5 in /opt/venv/lib64/python3.13/site-packages/torch/lib/libtorch_hip.so)
+frame #4: + 0x4e3e4 (0x7f01dce603e4 in /lib64/libstdc++.so.6)
+frame #5: + 0x72464 (0x7f01dd0fb464 in /lib64/libc.so.6)
+frame #6: + 0xf55ac (0x7f01dd17e5ac in /lib64/libc.so.6)
+
+terminate called after throwing an instance of 'c10::DistBackendError'
+ what(): [PG ID 2 PG GUID 3 Rank 1] Process group watchdog thread terminated with exception: HIP error: an illegal memory access was encountered
+Search for `hipErrorIllegalAddress' in https://rocm.docs.amd.com/projects/HIP/en/latest/index.html for more information.
+HIP kernel errors might be asynchronously reported at some other API call, so the stacktrace below might be incorrect.
+For debugging consider passing AMD_SERIALIZE_KERNEL=3
+Compile with `TORCH_USE_HIP_DSA` to enable device-side assertions.
+
+Exception raised from getDevice at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/hip/impl/HIPGuardImplMasqueradingAsCUDA.h:75 (most recent call first):
+frame #0: c10::Error::Error(c10::SourceLocation, std::__cxx11::basic_string, std::allocator >) + 0x9d (0x7f01713fafdd in /opt/venv/lib64/python3.13/site-packages/torch/lib/libc10.so)
+frame #1: + 0x12568 (0x7f0171486568 in /opt/venv/lib64/python3.13/site-packages/torch/lib/libc10_hip.so)
+frame #2: c10d::ProcessGroupNCCL::Watchdog::runLoop() + 0x922 (0x7f0174a7ccc2 in /opt/venv/lib64/python3.13/site-packages/torch/lib/libtorch_hip.so)
+frame #3: c10d::ProcessGroupNCCL::Watchdog::run() + 0x105 (0x7f0174a7f1b5 in /opt/venv/lib64/python3.13/site-packages/torch/lib/libtorch_hip.so)
+frame #4: + 0x4e3e4 (0x7f01dce603e4 in /lib64/libstdc++.so.6)
+frame #5: + 0x72464 (0x7f01dd0fb464 in /lib64/libc.so.6)
+frame #6: + 0xf55ac (0x7f01dd17e5ac in /lib64/libc.so.6)
+
+Exception raised from run at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/torch/csrc/distributed/c10d/ProcessGroupNCCL.cpp:2099 (most recent call first):
+frame #0: c10::Error::Error(c10::SourceLocation, std::__cxx11::basic_string, std::allocator >) + 0x9d (0x7f01713fafdd in /opt/venv/lib64/python3.13/site-packages/torch/lib/libc10.so)
+frame #1: + 0xb51c0c (0x7f0172129c0c in /opt/venv/lib64/python3.13/site-packages/torch/lib/libtorch_hip.so)
+frame #2: + 0x4e3e4 (0x7f01dce603e4 in /lib64/libstdc++.so.6)
+frame #3: + 0x72464 (0x7f01dd0fb464 in /lib64/libc.so.6)
+frame #4: + 0xf55ac (0x7f01dd17e5ac in /lib64/libc.so.6)
+
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:60172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:16 [multiproc_executor.py:233] Worker proc VllmWorker-1 died unexpectedly, shutting down executor.
+[0;36m(Worker_TP0 pid=77933)[0;0m INFO 12-19 17:38:16 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:59124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:59126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO 12-19 17:38:18 [loggers.py:248] Engine 000: Avg prompt throughput: 172.0 tokens/s, Avg generation throughput: 150.8 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:59134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=77686)[0;0m INFO: 127.0.0.1:59138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [dump_input.py:72] Dumping input data for V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='RedHatAI/gemma-3-27b-it-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/gemma-3-27b-it-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=28900, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/gemma-3-27b-it-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 64, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None},
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [dump_input.py:79] Dumping scheduler output for model execution: SchedulerOutput(scheduled_new_reqs=[NewRequestData(req_id=cmpl-bench-0a333e52-45-0,prompt_token_ids_len=282,mm_features=[],sampling_params=SamplingParams(n=1, presence_penalty=0.0, frequency_penalty=0.0, repetition_penalty=1.0, temperature=0.0, top_p=1.0, top_k=0, min_p=0.0, seed=None, stop=[], stop_token_ids=[106], bad_words=[], include_stop_str_in_output=False, ignore_eos=False, max_tokens=493, min_tokens=0, logprobs=None, prompt_logprobs=None, skip_special_tokens=True, spaces_between_special_tokens=True, truncate_prompt_tokens=None, structured_outputs=None, extra_args=None),block_ids=([7974, 7975, 7976, 7977, 7978, 7979, 7980, 7981, 7982, 7983, 7984, 7985, 7986, 7987, 7988, 7989, 7990, 7991], [7992, 7993, 7994, 7995, 7996, 7997, 7998, 7999, 8000, 8001, 8002, 8003, 8004, 8005, 8006, 8007, 8008, 8009], [8010, 8011, 8012, 8013, 8014, 8015, 8016, 8017, 8018, 8019, 8020, 8021, 8022, 8023, 8024, 8025, 8026, 8027], [8028, 8029, 8030, 8031, 8032, 8033, 8034, 8035, 8036, 8037, 8038, 8039, 8040, 8041, 8042, 8043, 8044, 8045], [8046, 8047, 8048, 8049, 8050, 8051, 8052, 8053, 8054, 8055, 8056, 8057, 8058, 8059, 8060, 8061, 8062, 8063], [8064, 8065, 8066, 8067, 8068, 8069, 8070, 8071, 8072, 8073, 8074, 8075, 8076, 8077, 8078, 8079, 8080, 8081], [8082, 8083, 8084, 8085, 8086, 8087, 8088, 8089, 8090, 8091, 8092, 8093, 8094, 8095, 8096, 8097, 8098, 8099]),num_computed_tokens=0,lora_request=None,prompt_embeds_shape=None)], scheduled_cached_reqs=CachedRequestData(req_ids=['cmpl-bench-0a333e52-5-0', 'cmpl-bench-0a333e52-20-0', 'cmpl-bench-0a333e52-21-0', 'cmpl-bench-0a333e52-26-0', 'cmpl-bench-0a333e52-34-0', 'cmpl-bench-0a333e52-35-0', 'cmpl-bench-0a333e52-36-0', 'cmpl-bench-0a333e52-39-0', 'cmpl-bench-0a333e52-40-0', 'cmpl-bench-0a333e52-43-0', 'cmpl-bench-0a333e52-44-0'], resumed_req_ids=[], new_token_ids=[], all_token_ids={}, new_block_ids=[null, null, null, null, null, null, null, null, null, null, null], num_computed_tokens=[877, 1009, 430, 726, 538, 213, 838, 87, 79, 47, 27], num_output_tokens=[849, 401, 372, 323, 184, 177, 159, 72, 63, 36, 21]), num_scheduled_tokens={cmpl-bench-0a333e52-40-0: 1, cmpl-bench-0a333e52-43-0: 1, cmpl-bench-0a333e52-44-0: 1, cmpl-bench-0a333e52-45-0: 282, cmpl-bench-0a333e52-39-0: 1, cmpl-bench-0a333e52-20-0: 1, cmpl-bench-0a333e52-5-0: 1, cmpl-bench-0a333e52-26-0: 1, cmpl-bench-0a333e52-21-0: 1, cmpl-bench-0a333e52-34-0: 1, cmpl-bench-0a333e52-36-0: 1, cmpl-bench-0a333e52-35-0: 1}, total_num_scheduled_tokens=293, scheduled_spec_decode_tokens={}, scheduled_encoder_inputs={}, num_common_prefix_blocks=[0, 0, 0, 0, 0, 0, 0], finished_req_ids=[], free_encoder_mm_hashes=[], preempted_req_ids=[], pending_structured_output_tokens=false, kv_connector_metadata=null, ec_connector_metadata=null)
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [dump_input.py:81] Dumping scheduler stats: SchedulerStats(num_running_reqs=12, num_waiting_reqs=0, step_counter=0, current_wave=0, kv_cache_usage=0.09605539236256821, prefix_cache_stats=PrefixCacheStats(reset=False, requests=1, queries=282, hits=0, preempted_requests=0, preempted_queries=0, preempted_hits=0), connector_prefix_cache_stats=None, kv_cache_eviction_events=[], spec_decoding_stats=None, kv_connector_stats=None, waiting_lora_adapters={}, running_lora_adapters={}, cudagraph_stats=None)
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] EngineCore encountered a fatal error.
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 859, in run_engine_core
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] engine_core.run_busy_loop()
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 886, in run_busy_loop
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] self._process_engine_step()
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 919, in _process_engine_step
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] outputs, model_executed = self.step_fn()
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 353, in step
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] model_output = self.model_executor.sample_tokens(grammar_output)
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 271, in sample_tokens
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] return self.collective_rpc(
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] "sample_tokens",
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ...<4 lines>...
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] kv_output_aggregator=self.kv_output_aggregator,
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] )
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 338, in get_response
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] status, result = mq.dequeue(
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] timeout=dequeue_timeout, cancel=shutdown_event
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] )
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/distributed/device_communicators/shm_broadcast.py", line 616, in dequeue
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] with self.acquire_read(timeout, cancel, indefinite) as buf:
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] ~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/usr/lib64/python3.13/contextlib.py", line 141, in __enter__
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] return next(self.gen)
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] File "/opt/venv/lib64/python3.13/site-packages/vllm/distributed/device_communicators/shm_broadcast.py", line 531, in acquire_read
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] raise RuntimeError("cancelled")
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ERROR 12-19 17:38:20 [core.py:868] RuntimeError: cancelled
+[0;36m(EngineCore_DP0 pid=77851)[0;0m Process EngineCore_DP0:
+[0;36m(EngineCore_DP0 pid=77851)[0;0m Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 313, in _bootstrap
+[0;36m(EngineCore_DP0 pid=77851)[0;0m self.run()
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 108, in run
+[0;36m(EngineCore_DP0 pid=77851)[0;0m self._target(*self._args, **self._kwargs)
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 870, in run_engine_core
+[0;36m(EngineCore_DP0 pid=77851)[0;0m raise e
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 859, in run_engine_core
+[0;36m(EngineCore_DP0 pid=77851)[0;0m engine_core.run_busy_loop()
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 886, in run_busy_loop
+[0;36m(EngineCore_DP0 pid=77851)[0;0m self._process_engine_step()
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 919, in _process_engine_step
+[0;36m(EngineCore_DP0 pid=77851)[0;0m outputs, model_executed = self.step_fn()
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 353, in step
+[0;36m(EngineCore_DP0 pid=77851)[0;0m model_output = self.model_executor.sample_tokens(grammar_output)
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 271, in sample_tokens
+[0;36m(EngineCore_DP0 pid=77851)[0;0m return self.collective_rpc(
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m "sample_tokens",
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ...<4 lines>...
+[0;36m(EngineCore_DP0 pid=77851)[0;0m kv_output_aggregator=self.kv_output_aggregator,
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m )
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=77851)[0;0m return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 338, in get_response
+[0;36m(EngineCore_DP0 pid=77851)[0;0m status, result = mq.dequeue(
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m timeout=dequeue_timeout, cancel=shutdown_event
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m )
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/distributed/device_communicators/shm_broadcast.py", line 616, in dequeue
+[0;36m(EngineCore_DP0 pid=77851)[0;0m with self.acquire_read(timeout, cancel, indefinite) as buf:
+[0;36m(EngineCore_DP0 pid=77851)[0;0m ~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 141, in __enter__
+[0;36m(EngineCore_DP0 pid=77851)[0;0m return next(self.gen)
+[0;36m(EngineCore_DP0 pid=77851)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/distributed/device_communicators/shm_broadcast.py", line 531, in acquire_read
+[0;36m(EngineCore_DP0 pid=77851)[0;0m raise RuntimeError("cancelled")
+[0;36m(EngineCore_DP0 pid=77851)[0;0m RuntimeError: cancelled
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] AsyncLLM output_handler failed.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [async_llm.py:538] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Error in completion stream generator.
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] Traceback (most recent call last):
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/serving_completion.py", line 352, in completion_stream_generator
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for prompt_idx, res in result_generator:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ...<125 lines>...
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield f"data: {response_json}\n\n"
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/utils/async_utils.py", line 278, in merge_async_iterators
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] async for item in iterators[0]:
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] yield 0, item
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 436, in generate
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] out = q.get_nowait() or await q.get()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/output_processor.py", line 70, in get
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise output
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 490, in output_handler
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] outputs = await engine_core.get_output_async()
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 895, in get_output_async
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] raise self._format_exception(outputs) from None
+[0;36m(APIServer pid=77686)[0;0m ERROR 12-19 17:38:20 [serving_completion.py:513] vllm.v1.engine.exceptions.EngineDeadError: EngineCore encountered an issue. See stack trace (above) for the root cause.
+[0;36m(APIServer pid=77686)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=77686)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=77686)[0;0m INFO: Application shutdown complete.
+[0;36m(APIServer pid=77686)[0;0m INFO: Finished server process [77686]
+/usr/lib64/python3.13/multiprocessing/resource_tracker.py:324: UserWarning: resource_tracker: There appear to be 2 leaked semaphore objects to clean up at shutdown: {'/mp-zs3r7w6q', '/mp-sdlfr_2q'}
+ warnings.warn(
+/usr/lib64/python3.13/multiprocessing/resource_tracker.py:324: UserWarning: resource_tracker: There appear to be 3 leaked shared_memory objects to clean up at shutdown: {'/psm_0f0e988c', '/psm_760784ea', '/psm_817cd347'}
+ warnings.warn(
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json
new file mode 100644
index 0000000..0be91c1
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 1215.8245900149996,
+ "num_requests": 1000,
+ "total_num_tokens": 755432,
+ "requests_per_second": 0.8224870661545537,
+ "tokens_per_second": 621.3330493592667
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..3a99a6b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 15:19:35 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-eb7743a2-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 200.40 \nTotal input tokens: 38358 \nTotal generated tokens: 40237 \nRequest throughput (req/s): 0.90 \nOutput token throughput (tok/s): 200.79 \nPeak output token throughput (tok/s): 315.00 \nPeak concurrent requests: 19.00 \nTotal token throughput (tok/s): 392.20 \n---------------Time to First Token----------------\nMean TTFT (ms): 125.43 \nMedian TTFT (ms): 115.75 \nP99 TTFT (ms): 233.19 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 51.71 \nMedian TPOT (ms): 52.62 \nP99 TPOT (ms): 60.97 \n---------------Inter-token Latency----------------\nMean ITL (ms): 51.22 \nMedian ITL (ms): 51.73 \nP99 ITL (ms): 115.89 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..e395862
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 15:23:04 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-783b8bf2-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 249.51 \nTotal input tokens: 146694 \nTotal generated tokens: 152753 \nRequest throughput (req/s): 2.89 \nOutput token throughput (tok/s): 612.20 \nPeak output token throughput (tok/s): 832.00 \nPeak concurrent requests: 171.00 \nTotal token throughput (tok/s): 1200.12 \n---------------Time to First Token----------------\nMean TTFT (ms): 14305.88 \nMedian TTFT (ms): 18966.25 \nP99 TTFT (ms): 31290.19 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 88.19 \nMedian TPOT (ms): 90.39 \nP99 TPOT (ms): 108.34 \n---------------Inter-token Latency----------------\nMean ITL (ms): 87.87 \nMedian ITL (ms): 84.16 \nP99 ITL (ms): 193.01 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
new file mode 100644
index 0000000..69bb2a4
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
@@ -0,0 +1,1038 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 15:18:51 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:18:51 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:18:51 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=59832)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:18:55 [model.py:514] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:18:55 [model.py:1636] Using max model len 24576
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:18:55 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:18:59 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:18:59 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:34261 backend=nccl
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:18:59 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:00 [gpu_model_runner.py:3562] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:00 [compressed_tensors_wNa16.py:114] Using ConchLinearKernel for CompressedTensorsWNA16
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:00 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:00 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:00 [compressed_tensors_moe.py:188] Using CompressedTensorsWNA16MoEMethod
+[0;36m(EngineCore_DP0 pid=59999)[0;0m WARNING 12-19 15:19:00 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:02, 1.43it/s]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:03<00:03, 1.75s/it]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:05<00:02, 2.03s/it]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:08<00:00, 2.25s/it]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:08<00:00, 2.03s/it]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:10 [default_loader.py:308] Loading weights took 8.18 seconds
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:10 [gpu_model_runner.py:3659] Model loading took 16.2266 GiB memory and 9.521962 seconds
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:15 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/8582846099/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:15 [backends.py:694] Dynamo bytecode transform time: 5.05 s
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:18 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=59999)[0;0m WARNING 12-19 15:19:19 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:21 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 2.44 s
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:21 [monitor.py:34] torch.compile takes 7.50 s in total
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:22 [gpu_worker.py:375] Available KV cache memory: 14.08 GiB
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:23 [kv_cache_utils.py:1291] GPU KV cache size: 153,776 tokens
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:23 [kv_cache_utils.py:1296] Maximum concurrency for 24,576 tokens per request: 6.26x
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 18.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 20.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 21.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 22.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 22.87it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 23.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 22.41it/s]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 6.03it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 17.01it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 23.71it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 22.70it/s]
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:25 [gpu_model_runner.py:4610] Graph capturing finished in 2 secs, took 1.14 GiB
+[0;36m(EngineCore_DP0 pid=59999)[0;0m INFO 12-19 15:19:25 [core.py:259] init engine (profile, create kv cache, warmup model) took 14.56 seconds
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=59832)[0;0m WARNING 12-19 15:19:26 [model.py:1462] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [serving_responses.py:201] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [serving_chat.py:137] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [serving_completion.py:77] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [serving_chat.py:137] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:26 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=59832)[0;0m INFO: Started server process [59832]
+[0;36m(APIServer pid=59832)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=59832)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58088 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:47 [loggers.py:248] Engine 000: Avg prompt throughput: 84.0 tokens/s, Avg generation throughput: 68.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:19:57 [loggers.py:248] Engine 000: Avg prompt throughput: 136.8 tokens/s, Avg generation throughput: 119.8 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:20:07 [loggers.py:248] Engine 000: Avg prompt throughput: 207.4 tokens/s, Avg generation throughput: 166.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:20:17 [loggers.py:248] Engine 000: Avg prompt throughput: 300.9 tokens/s, Avg generation throughput: 182.7 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:20:27 [loggers.py:248] Engine 000: Avg prompt throughput: 259.1 tokens/s, Avg generation throughput: 176.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:20:37 [loggers.py:248] Engine 000: Avg prompt throughput: 251.4 tokens/s, Avg generation throughput: 213.0 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:20:47 [loggers.py:248] Engine 000: Avg prompt throughput: 488.7 tokens/s, Avg generation throughput: 231.2 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:20:57 [loggers.py:248] Engine 000: Avg prompt throughput: 149.9 tokens/s, Avg generation throughput: 198.1 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:21:07 [loggers.py:248] Engine 000: Avg prompt throughput: 299.2 tokens/s, Avg generation throughput: 231.5 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:21:17 [loggers.py:248] Engine 000: Avg prompt throughput: 310.1 tokens/s, Avg generation throughput: 272.8 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:21:27 [loggers.py:248] Engine 000: Avg prompt throughput: 132.9 tokens/s, Avg generation throughput: 285.6 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:21:37 [loggers.py:248] Engine 000: Avg prompt throughput: 109.9 tokens/s, Avg generation throughput: 262.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:21:47 [loggers.py:248] Engine 000: Avg prompt throughput: 248.5 tokens/s, Avg generation throughput: 215.5 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:21:57 [loggers.py:248] Engine 000: Avg prompt throughput: 182.1 tokens/s, Avg generation throughput: 243.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:57090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:22:07 [loggers.py:248] Engine 000: Avg prompt throughput: 130.1 tokens/s, Avg generation throughput: 217.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:22:17 [loggers.py:248] Engine 000: Avg prompt throughput: 80.4 tokens/s, Avg generation throughput: 154.9 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:58042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:22:27 [loggers.py:248] Engine 000: Avg prompt throughput: 140.2 tokens/s, Avg generation throughput: 203.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:32832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:51614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:22:37 [loggers.py:248] Engine 000: Avg prompt throughput: 275.0 tokens/s, Avg generation throughput: 184.4 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:22:47 [loggers.py:248] Engine 000: Avg prompt throughput: 50.4 tokens/s, Avg generation throughput: 228.4 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:22:57 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 147.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:23:07 [loggers.py:248] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 35.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:23:17 [loggers.py:248] Engine 000: Avg prompt throughput: 641.4 tokens/s, Avg generation throughput: 192.3 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 13.8%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:23:27 [loggers.py:248] Engine 000: Avg prompt throughput: 1213.2 tokens/s, Avg generation throughput: 460.5 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.7%, Prefix cache hit rate: 31.4%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:23:37 [loggers.py:248] Engine 000: Avg prompt throughput: 797.2 tokens/s, Avg generation throughput: 589.9 tokens/s, Running: 52 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.4%, Prefix cache hit rate: 39.2%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:23:47 [loggers.py:248] Engine 000: Avg prompt throughput: 674.7 tokens/s, Avg generation throughput: 685.8 tokens/s, Running: 61 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.6%, Prefix cache hit rate: 44.6%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:23:57 [loggers.py:248] Engine 000: Avg prompt throughput: 593.0 tokens/s, Avg generation throughput: 726.2 tokens/s, Running: 64 reqs, Waiting: 11 reqs, GPU KV cache usage: 16.5%, Prefix cache hit rate: 47.5%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:24:07 [loggers.py:248] Engine 000: Avg prompt throughput: 542.8 tokens/s, Avg generation throughput: 684.8 tokens/s, Running: 61 reqs, Waiting: 28 reqs, GPU KV cache usage: 16.1%, Prefix cache hit rate: 44.4%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:24:17 [loggers.py:248] Engine 000: Avg prompt throughput: 920.6 tokens/s, Avg generation throughput: 665.5 tokens/s, Running: 64 reqs, Waiting: 25 reqs, GPU KV cache usage: 16.3%, Prefix cache hit rate: 39.9%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:24:27 [loggers.py:248] Engine 000: Avg prompt throughput: 565.0 tokens/s, Avg generation throughput: 678.4 tokens/s, Running: 64 reqs, Waiting: 42 reqs, GPU KV cache usage: 16.9%, Prefix cache hit rate: 37.6%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:24:37 [loggers.py:248] Engine 000: Avg prompt throughput: 444.7 tokens/s, Avg generation throughput: 710.4 tokens/s, Running: 64 reqs, Waiting: 58 reqs, GPU KV cache usage: 14.9%, Prefix cache hit rate: 36.0%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:24:47 [loggers.py:248] Engine 000: Avg prompt throughput: 384.4 tokens/s, Avg generation throughput: 729.6 tokens/s, Running: 64 reqs, Waiting: 73 reqs, GPU KV cache usage: 14.1%, Prefix cache hit rate: 34.7%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:24:57 [loggers.py:248] Engine 000: Avg prompt throughput: 614.0 tokens/s, Avg generation throughput: 697.6 tokens/s, Running: 64 reqs, Waiting: 81 reqs, GPU KV cache usage: 13.4%, Prefix cache hit rate: 32.8%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:25:07 [loggers.py:248] Engine 000: Avg prompt throughput: 690.7 tokens/s, Avg generation throughput: 697.5 tokens/s, Running: 64 reqs, Waiting: 73 reqs, GPU KV cache usage: 12.4%, Prefix cache hit rate: 30.9%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:25:17 [loggers.py:248] Engine 000: Avg prompt throughput: 935.6 tokens/s, Avg generation throughput: 659.2 tokens/s, Running: 64 reqs, Waiting: 74 reqs, GPU KV cache usage: 14.5%, Prefix cache hit rate: 28.6%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:25:27 [loggers.py:248] Engine 000: Avg prompt throughput: 687.6 tokens/s, Avg generation throughput: 684.7 tokens/s, Running: 64 reqs, Waiting: 85 reqs, GPU KV cache usage: 13.3%, Prefix cache hit rate: 27.2%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:39982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:25:37 [loggers.py:248] Engine 000: Avg prompt throughput: 750.4 tokens/s, Avg generation throughput: 671.9 tokens/s, Running: 63 reqs, Waiting: 83 reqs, GPU KV cache usage: 13.7%, Prefix cache hit rate: 25.8%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:38918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:38924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:38938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:38944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:25:47 [loggers.py:248] Engine 000: Avg prompt throughput: 554.3 tokens/s, Avg generation throughput: 697.6 tokens/s, Running: 62 reqs, Waiting: 83 reqs, GPU KV cache usage: 13.9%, Prefix cache hit rate: 24.9%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:25:57 [loggers.py:248] Engine 000: Avg prompt throughput: 909.9 tokens/s, Avg generation throughput: 665.6 tokens/s, Running: 64 reqs, Waiting: 81 reqs, GPU KV cache usage: 13.2%, Prefix cache hit rate: 23.5%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:53364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:34874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:44286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:43672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:40040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:26:07 [loggers.py:248] Engine 000: Avg prompt throughput: 584.0 tokens/s, Avg generation throughput: 691.2 tokens/s, Running: 63 reqs, Waiting: 100 reqs, GPU KV cache usage: 14.5%, Prefix cache hit rate: 22.7%
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:38944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:41430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:26:17 [loggers.py:248] Engine 000: Avg prompt throughput: 523.2 tokens/s, Avg generation throughput: 710.4 tokens/s, Running: 63 reqs, Waiting: 84 reqs, GPU KV cache usage: 13.6%, Prefix cache hit rate: 22.0%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:26:27 [loggers.py:248] Engine 000: Avg prompt throughput: 762.8 tokens/s, Avg generation throughput: 704.0 tokens/s, Running: 64 reqs, Waiting: 52 reqs, GPU KV cache usage: 15.5%, Prefix cache hit rate: 21.0%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:26:37 [loggers.py:248] Engine 000: Avg prompt throughput: 734.4 tokens/s, Avg generation throughput: 704.0 tokens/s, Running: 64 reqs, Waiting: 13 reqs, GPU KV cache usage: 14.0%, Prefix cache hit rate: 20.2%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:26:47 [loggers.py:248] Engine 000: Avg prompt throughput: 144.9 tokens/s, Avg generation throughput: 714.5 tokens/s, Running: 50 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:26:57 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 503.9 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.9%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:27:07 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 238.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:27:17 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 113.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=59832)[0;0m INFO 12-19 15:27:18 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1219 15:27:18.582580780 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
new file mode 100644
index 0000000..f24c281
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 723.7066626509986,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 1.3817753125788776,
+ "tokens_per_second": 1024.3570195753498
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..4412622
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 16:03:56 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-2904e4f8-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 192.44 \nTotal input tokens: 38358 \nTotal generated tokens: 39896 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 207.32 \nPeak output token throughput (tok/s): 372.00 \nPeak concurrent requests: 15.00 \nTotal token throughput (tok/s): 406.64 \n---------------Time to First Token----------------\nMean TTFT (ms): 97.06 \nMedian TTFT (ms): 80.47 \nP99 TTFT (ms): 208.68 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 32.90 \nMedian TPOT (ms): 32.05 \nP99 TPOT (ms): 41.21 \n---------------Inter-token Latency----------------\nMean ITL (ms): 32.45 \nMedian ITL (ms): 30.18 \nP99 ITL (ms): 85.86 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..3af696c
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 16:53:26 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-c9db55f5-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 206.61 \nTotal input tokens: 146694 \nTotal generated tokens: 153351 \nRequest throughput (req/s): 3.48 \nOutput token throughput (tok/s): 742.22 \nPeak output token throughput (tok/s): 1152.00 \nPeak concurrent requests: 76.00 \nTotal token throughput (tok/s): 1452.22 \n---------------Time to First Token----------------\nMean TTFT (ms): 194.04 \nMedian TTFT (ms): 127.11 \nP99 TTFT (ms): 1506.44 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 58.93 \nMedian TPOT (ms): 59.18 \nP99 TPOT (ms): 82.75 \n---------------Inter-token Latency----------------\nMean ITL (ms): 58.36 \nMedian ITL (ms): 52.92 \nP99 ITL (ms): 178.94 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log
new file mode 100644
index 0000000..755a6a8
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log
@@ -0,0 +1,846 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 16:52:23 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:52:23 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:52:23 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=69653)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:52:28 [model.py:514] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:52:28 [model.py:1636] Using max model len 24576
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:52:28 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=69815)[0;0m INFO 12-19 16:52:32 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=69815)[0;0m WARNING 12-19 16:52:32 [multiproc_executor.py:884] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+INFO 12-19 16:52:35 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:49789 backend=nccl
+INFO 12-19 16:52:35 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:49789 backend=nccl
+INFO 12-19 16:52:35 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-19 16:52:36 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+INFO 12-19 16:52:36 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:36 [gpu_model_runner.py:3562] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(Worker_TP1 pid=69899)[0;0m INFO 12-19 16:52:37 [compressed_tensors_wNa16.py:114] Using ConchLinearKernel for CompressedTensorsWNA16
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:37 [compressed_tensors_wNa16.py:114] Using ConchLinearKernel for CompressedTensorsWNA16
+[0;36m(Worker_TP1 pid=69899)[0;0m INFO 12-19 16:52:37 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=69899)[0;0m INFO 12-19 16:52:37 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP1 pid=69899)[0;0m INFO 12-19 16:52:37 [compressed_tensors_moe.py:188] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP1 pid=69899)[0;0m WARNING 12-19 16:52:37 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:37 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:37 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:37 [compressed_tensors_moe.py:188] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP0 pid=69898)[0;0m WARNING 12-19 16:52:37 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=69898)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=69898)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:01, 1.61it/s]
+[0;36m(Worker_TP0 pid=69898)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:02<00:03, 1.60s/it]
+[0;36m(Worker_TP0 pid=69898)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:05<00:01, 1.90s/it]
+[0;36m(Worker_TP0 pid=69898)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:07<00:00, 2.13s/it]
+[0;36m(Worker_TP0 pid=69898)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:07<00:00, 1.91s/it]
+[0;36m(Worker_TP0 pid=69898)[0;0m
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:46 [default_loader.py:308] Loading weights took 7.71 seconds
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:46 [gpu_model_runner.py:3659] Model loading took 8.1992 GiB memory and 9.032124 seconds
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:52 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/48c10e1f97/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:52 [backends.py:694] Dynamo bytecode transform time: 5.46 s
+[0;36m(Worker_TP1 pid=69899)[0;0m INFO 12-19 16:52:55 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:55 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP1 pid=69899)[0;0m WARNING 12-19 16:52:55 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP0 pid=69898)[0;0m WARNING 12-19 16:52:55 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:57 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 2.62 s
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:52:57 [monitor.py:34] torch.compile takes 8.08 s in total
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:53:00 [gpu_worker.py:375] Available KV cache memory: 22.22 GiB
+[0;36m(EngineCore_DP0 pid=69815)[0;0m INFO 12-19 16:53:00 [kv_cache_utils.py:1291] GPU KV cache size: 485,344 tokens
+[0;36m(EngineCore_DP0 pid=69815)[0;0m INFO 12-19 16:53:00 [kv_cache_utils.py:1296] Maximum concurrency for 24,576 tokens per request: 19.75x
+[0;36m(Worker_TP0 pid=69898)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:09, 1.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:01<00:08, 1.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:01<00:08, 1.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:01<00:07, 2.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:02<00:06, 2.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:02<00:06, 2.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:03<00:05, 2.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:03<00:05, 2.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:04<00:04, 2.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:04<00:04, 2.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:05<00:04, 1.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:06<00:03, 1.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:06<00:03, 1.94it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:07<00:02, 1.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:07<00:02, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:08<00:01, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:08<00:01, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:09<00:00, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:09<00:00, 1.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:09<00:00, 1.97it/s]
+[0;36m(Worker_TP0 pid=69898)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:05, 1.75it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:04, 2.07it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:01<00:03, 2.18it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:01<00:03, 2.24it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:02<00:02, 2.28it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:02<00:02, 2.33it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:03<00:01, 2.34it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:03<00:01, 2.34it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:03<00:00, 2.33it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:04<00:00, 2.33it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.34it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.28it/s]
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:53:16 [gpu_model_runner.py:4610] Graph capturing finished in 15 secs, took 0.76 GiB
+[0;36m(EngineCore_DP0 pid=69815)[0;0m INFO 12-19 16:53:16 [core.py:259] init engine (profile, create kv cache, warmup model) took 29.33 seconds
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=69653)[0;0m WARNING 12-19 16:53:17 [model.py:1462] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [serving_responses.py:201] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [serving_chat.py:137] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [serving_completion.py:77] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [serving_chat.py:137] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:17 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=69653)[0;0m INFO: Started server process [69653]
+[0;36m(APIServer pid=69653)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=69653)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:36624 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:38 [loggers.py:248] Engine 000: Avg prompt throughput: 428.2 tokens/s, Avg generation throughput: 153.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:48 [loggers.py:248] Engine 000: Avg prompt throughput: 1264.2 tokens/s, Avg generation throughput: 435.2 tokens/s, Running: 38 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:53:58 [loggers.py:248] Engine 000: Avg prompt throughput: 925.9 tokens/s, Avg generation throughput: 700.7 tokens/s, Running: 41 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:54:08 [loggers.py:248] Engine 000: Avg prompt throughput: 643.5 tokens/s, Avg generation throughput: 815.1 tokens/s, Running: 54 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:54:18 [loggers.py:248] Engine 000: Avg prompt throughput: 662.7 tokens/s, Avg generation throughput: 855.4 tokens/s, Running: 51 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:54:28 [loggers.py:248] Engine 000: Avg prompt throughput: 978.9 tokens/s, Avg generation throughput: 826.0 tokens/s, Running: 53 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:54:38 [loggers.py:248] Engine 000: Avg prompt throughput: 936.6 tokens/s, Avg generation throughput: 896.8 tokens/s, Running: 58 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:54:48 [loggers.py:248] Engine 000: Avg prompt throughput: 480.3 tokens/s, Avg generation throughput: 907.4 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:54:58 [loggers.py:248] Engine 000: Avg prompt throughput: 919.2 tokens/s, Avg generation throughput: 935.0 tokens/s, Running: 63 reqs, Waiting: 5 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:55:08 [loggers.py:248] Engine 000: Avg prompt throughput: 627.6 tokens/s, Avg generation throughput: 1008.0 tokens/s, Running: 55 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:55:18 [loggers.py:248] Engine 000: Avg prompt throughput: 1008.7 tokens/s, Avg generation throughput: 869.9 tokens/s, Running: 50 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.8%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:55:28 [loggers.py:248] Engine 000: Avg prompt throughput: 756.8 tokens/s, Avg generation throughput: 756.5 tokens/s, Running: 45 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 0.8%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:55:38 [loggers.py:248] Engine 000: Avg prompt throughput: 690.0 tokens/s, Avg generation throughput: 750.2 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.7%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:55:48 [loggers.py:248] Engine 000: Avg prompt throughput: 757.2 tokens/s, Avg generation throughput: 808.7 tokens/s, Running: 46 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.9%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:55:58 [loggers.py:248] Engine 000: Avg prompt throughput: 840.8 tokens/s, Avg generation throughput: 722.0 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.8%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:56:08 [loggers.py:248] Engine 000: Avg prompt throughput: 823.4 tokens/s, Avg generation throughput: 727.0 tokens/s, Running: 45 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.8%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:56:18 [loggers.py:248] Engine 000: Avg prompt throughput: 495.1 tokens/s, Avg generation throughput: 806.8 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.7%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:56:28 [loggers.py:248] Engine 000: Avg prompt throughput: 1132.0 tokens/s, Avg generation throughput: 769.2 tokens/s, Running: 47 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.7%
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:60354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:34056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:54814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:56926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO: 127.0.0.1:41560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:56:38 [loggers.py:248] Engine 000: Avg prompt throughput: 298.6 tokens/s, Avg generation throughput: 914.9 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.7%
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:56:48 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 512.4 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.7%
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:56:58 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 175.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.7%
+[0;36m(APIServer pid=69653)[0;0m INFO 12-19 16:56:58 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP1 pid=69899)[0;0m INFO 12-19 16:56:58 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(Worker_TP0 pid=69898)[0;0m INFO 12-19 16:56:58 [multiproc_executor.py:711] Parent process exited, terminating worker
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json
new file mode 100644
index 0000000..2fe5962
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 498.7602350669986,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 2.004971386433142,
+ "tokens_per_second": 1486.353457790027
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log
new file mode 100644
index 0000000..3378c00
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log
@@ -0,0 +1,941 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 17:11:45 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:45 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:45 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', 'trust_remote_code': True, 'max_model_len': 16384, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 32}
+[0;36m(APIServer pid=76251)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:49 [model.py:514] Resolved architecture: Qwen3NextForCausalLM
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:49 [model.py:1636] Using max model len 16384
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:49 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:49 [config.py:312] Disabling cascade attention since it is not supported for hybrid models.
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:49 [config.py:439] Setting attention block size to 544 tokens to ensure that attention page size is >= mamba page size.
+[0;36m(APIServer pid=76251)[0;0m INFO 12-19 17:11:49 [config.py:463] Padding mamba page size by 1.49% to ensure that mamba page size and attention page size are exactly equal.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=76413)[0;0m INFO 12-19 17:11:53 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=16384, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit, enable_prefix_caching=False, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 64, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=76413)[0;0m WARNING 12-19 17:11:53 [multiproc_executor.py:884] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+INFO 12-19 17:11:56 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:50157 backend=nccl
+INFO 12-19 17:11:56 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:50157 backend=nccl
+INFO 12-19 17:11:57 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-19 17:11:57 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+INFO 12-19 17:11:57 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:11:58 [gpu_model_runner.py:3562] Starting to load model cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit...
+[0;36m(Worker_TP1 pid=76496)[0;0m WARNING 12-19 17:11:58 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP1 pid=76496)[0;0m INFO 12-19 17:11:58 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP1 pid=76496)[0;0m INFO 12-19 17:11:58 [compressed_tensors_moe.py:188] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP0 pid=76495)[0;0m WARNING 12-19 17:11:58 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:11:58 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:11:58 [compressed_tensors_moe.py:188] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP1 pid=76496)[0;0m INFO 12-19 17:11:58 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:11:58 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/10 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 10% Completed | 1/10 [00:03<00:32, 3.65s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 20% Completed | 2/10 [00:07<00:29, 3.64s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 30% Completed | 3/10 [00:11<00:26, 3.79s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 40% Completed | 4/10 [00:15<00:23, 3.97s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 5/10 [00:18<00:18, 3.76s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 60% Completed | 6/10 [00:22<00:15, 3.85s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 70% Completed | 7/10 [00:26<00:11, 3.83s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 80% Completed | 8/10 [00:30<00:07, 3.84s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 10/10 [00:34<00:00, 2.97s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 10/10 [00:34<00:00, 3.46s/it]
+[0;36m(Worker_TP0 pid=76495)[0;0m
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:33 [default_loader.py:308] Loading weights took 34.64 seconds
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:34 [gpu_model_runner.py:3659] Model loading took 23.5020 GiB memory and 35.701153 seconds
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:38 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/671837fea6/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:38 [backends.py:694] Dynamo bytecode transform time: 4.08 s
+[0;36m(Worker_TP1 pid=76496)[0;0m INFO 12-19 17:12:40 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:40 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP1 pid=76496)[0;0m WARNING 12-19 17:12:41 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP0 pid=76495)[0;0m WARNING 12-19 17:12:41 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:43 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 2.59 s
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:43 [monitor.py:34] torch.compile takes 6.67 s in total
+[0;36m(Worker_TP1 pid=76496)[0;0m WARNING 12-19 17:12:43 [decorators.py:528] Cannot save aot compilation to path /home/kyuz0/.cache/vllm/torch_aot_compile/8cfc5b4a85b0195581f74785744c13daeda3dd4bc8885cc84b5001d538a63819/rank_1_0/model, error:
+[0;36m(Worker_TP0 pid=76495)[0;0m WARNING 12-19 17:12:43 [decorators.py:528] Cannot save aot compilation to path /home/kyuz0/.cache/vllm/torch_aot_compile/8cfc5b4a85b0195581f74785744c13daeda3dd4bc8885cc84b5001d538a63819/rank_0_0/model, error:
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:44 [gpu_worker.py:375] Available KV cache memory: 7.09 GiB
+[0;36m(EngineCore_DP0 pid=76413)[0;0m INFO 12-19 17:12:45 [kv_cache_utils.py:1291] GPU KV cache size: 154,496 tokens
+[0;36m(EngineCore_DP0 pid=76413)[0;0m INFO 12-19 17:12:45 [kv_cache_utils.py:1296] Maximum concurrency for 16,384 tokens per request: 33.47x
+[0;36m(Worker_TP0 pid=76495)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 9%|▉ | 1/11 [00:00<00:05, 1.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 2/11 [00:01<00:04, 1.83it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 3/11 [00:01<00:04, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▋ | 4/11 [00:02<00:03, 1.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▌ | 5/11 [00:02<00:03, 1.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▍ | 6/11 [00:03<00:02, 1.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▎ | 7/11 [00:03<00:02, 1.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 8/11 [00:04<00:01, 1.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 9/11 [00:04<00:01, 1.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 91%|█████████ | 10/11 [00:05<00:00, 1.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 1.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 1.95it/s]
+[0;36m(Worker_TP0 pid=76495)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] WorkerProc hit an exception.
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 43, in wrapper
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return fn(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 1638, in arange
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return _semantic.arange(start, end)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/semantic.py", line 583, in arange
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise ValueError("arange's range must be a power of 2")
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ValueError: arange's range must be a power of 2
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 459, in compile_or_warm_up_model
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cuda_graph_memory_bytes = self.model_runner.capture_model()
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4586, in capture_model
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._capture_cudagraphs(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] compilation_cases=compilation_cases_decode,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cudagraph_runtime_mode=CUDAGraphMode.FULL,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] uniform_decode=True,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4664, in _capture_cudagraphs
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._dummy_run(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_tokens,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<6 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] activate_lora=activate_lora,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4198, in _dummy_run
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] outputs = self.model(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids=input_ids,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] **model_kwargs,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/cuda_graph.py", line 220, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.runnable(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 1231, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] hidden_states = self.model(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids, positions, intermediate_tensors, inputs_embeds
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/decorators.py", line 376, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.aot_compiled_fn(self, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_dynamo/aot_compile.py", line 124, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.fn(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 997, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] def forward(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/caching.py", line 54, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.optimized_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".99", line 333, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] submod_7 = self.submod_7(getitem_21, s72, getitem_22, getitem_23, getitem_24); getitem_21 = getitem_22 = getitem_23 = submod_7 = None
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".107", line 5, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] unified_attention_with_output = torch.ops.vllm.unified_attention_with_output(query_8, key_8, value_9, output_5, 'model.layers.3.self_attn.attn'); query_8 = key_8 = value_9 = output_5 = unified_attention_with_output = None
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_ops.py", line 1209, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._op(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/utils/kv_transfer_utils.py", line 39, in wrapper
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/layer.py", line 923, in unified_attention_with_output
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self.impl.forward(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<7 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_block_scale=output_block_scale,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/attention/backends/rocm_attn.py", line 337, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] chunked_prefill_paged_decode(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] query=query[:num_actual_tokens],
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<17 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] sinks=self.sinks,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/ops/chunked_prefill_paged_decode.py", line 356, in chunked_prefill_paged_decode
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel_paged_attention_2d[
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ](
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_ptr=output,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<37 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] USE_FP8=output_scale is not None,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 419, in
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 733, in run
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self._do_compile(key, signature, device, constexprs, options, attrs, warmup)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 861, in _do_compile
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self.compile(src, target=target, options=options.__dict__)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 300, in compile
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module = src.make_ir(target, options, codegen_fns, module_map, context)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 80, in make_ir
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return ast_to_ttir(self.fn, self, context=context, options=options, codegen_fns=codegen_fns,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module_map=module_map)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] triton.compiler.errors.CompilationError: at 106:17:
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] if USE_ALIBI_SLOPES:
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slope = tl.load(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slopes_ptr + query_head_idx, mask=head_mask, other=0.0
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_blocks = cdiv_fn(seq_len, BLOCK_SIZE)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] # iterate through tiles
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] for j in range(0, num_blocks):
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] physical_block_idx = tl.load(block_tables_ptr + block_table_offset + j)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] offs_n = tl.arange(0, BLOCK_SIZE)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] arange's range must be a power of 2
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 43, in wrapper
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return fn(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 1638, in arange
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return _semantic.arange(start, end)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/semantic.py", line 583, in arange
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise ValueError("arange's range must be a power of 2")
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ValueError: arange's range must be a power of 2
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 459, in compile_or_warm_up_model
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cuda_graph_memory_bytes = self.model_runner.capture_model()
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4586, in capture_model
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._capture_cudagraphs(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] compilation_cases=compilation_cases_decode,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cudagraph_runtime_mode=CUDAGraphMode.FULL,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] uniform_decode=True,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4664, in _capture_cudagraphs
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._dummy_run(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_tokens,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<6 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] activate_lora=activate_lora,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4198, in _dummy_run
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] outputs = self.model(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids=input_ids,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] **model_kwargs,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/cuda_graph.py", line 220, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.runnable(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 1231, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] hidden_states = self.model(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids, positions, intermediate_tensors, inputs_embeds
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/decorators.py", line 376, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.aot_compiled_fn(self, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_dynamo/aot_compile.py", line 124, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.fn(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 997, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] def forward(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/caching.py", line 54, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.optimized_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".99", line 333, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] submod_7 = self.submod_7(getitem_21, s72, getitem_22, getitem_23, getitem_24); getitem_21 = getitem_22 = getitem_23 = submod_7 = None
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".107", line 5, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] unified_attention_with_output = torch.ops.vllm.unified_attention_with_output(query_8, key_8, value_9, output_5, 'model.layers.3.self_attn.attn'); query_8 = key_8 = value_9 = output_5 = unified_attention_with_output = None
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_ops.py", line 1209, in __call__
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._op(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/utils/kv_transfer_utils.py", line 39, in wrapper
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/layer.py", line 923, in unified_attention_with_output
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self.impl.forward(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<7 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_block_scale=output_block_scale,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/attention/backends/rocm_attn.py", line 337, in forward
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] chunked_prefill_paged_decode(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] query=query[:num_actual_tokens],
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<17 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] sinks=self.sinks,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/ops/chunked_prefill_paged_decode.py", line 356, in chunked_prefill_paged_decode
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel_paged_attention_2d[
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ](
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_ptr=output,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<37 lines>...
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] USE_FP8=output_scale is not None,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 419, in
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 733, in run
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self._do_compile(key, signature, device, constexprs, options, attrs, warmup)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 861, in _do_compile
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self.compile(src, target=target, options=options.__dict__)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 300, in compile
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module = src.make_ir(target, options, codegen_fns, module_map, context)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 80, in make_ir
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return ast_to_ttir(self.fn, self, context=context, options=options, codegen_fns=codegen_fns,
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module_map=module_map)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] triton.compiler.errors.CompilationError: at 106:17:
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] if USE_ALIBI_SLOPES:
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slope = tl.load(
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slopes_ptr + query_head_idx, mask=head_mask, other=0.0
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_blocks = cdiv_fn(seq_len, BLOCK_SIZE)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] # iterate through tiles
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] for j in range(0, num_blocks):
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] physical_block_idx = tl.load(block_tables_ptr + block_table_offset + j)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] offs_n = tl.arange(0, BLOCK_SIZE)
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] arange's range must be a power of 2
+[0;36m(Worker_TP1 pid=76496)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] WorkerProc hit an exception.
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 43, in wrapper
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return fn(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 1638, in arange
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return _semantic.arange(start, end)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/semantic.py", line 583, in arange
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise ValueError("arange's range must be a power of 2")
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ValueError: arange's range must be a power of 2
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 459, in compile_or_warm_up_model
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cuda_graph_memory_bytes = self.model_runner.capture_model()
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4586, in capture_model
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._capture_cudagraphs(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] compilation_cases=compilation_cases_decode,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cudagraph_runtime_mode=CUDAGraphMode.FULL,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] uniform_decode=True,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4664, in _capture_cudagraphs
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._dummy_run(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_tokens,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<6 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] activate_lora=activate_lora,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4198, in _dummy_run
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] outputs = self.model(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids=input_ids,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] **model_kwargs,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/cuda_graph.py", line 220, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.runnable(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 1231, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] hidden_states = self.model(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids, positions, intermediate_tensors, inputs_embeds
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/decorators.py", line 376, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.aot_compiled_fn(self, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_dynamo/aot_compile.py", line 124, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.fn(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 997, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] def forward(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/caching.py", line 54, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.optimized_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".99", line 333, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] submod_7 = self.submod_7(getitem_21, s72, getitem_22, getitem_23, getitem_24); getitem_21 = getitem_22 = getitem_23 = submod_7 = None
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".107", line 5, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] unified_attention_with_output = torch.ops.vllm.unified_attention_with_output(query_8, key_8, value_9, output_5, 'model.layers.3.self_attn.attn'); query_8 = key_8 = value_9 = output_5 = unified_attention_with_output = None
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_ops.py", line 1209, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._op(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/utils/kv_transfer_utils.py", line 39, in wrapper
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/layer.py", line 923, in unified_attention_with_output
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self.impl.forward(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<7 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_block_scale=output_block_scale,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/attention/backends/rocm_attn.py", line 337, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] chunked_prefill_paged_decode(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] query=query[:num_actual_tokens],
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<17 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] sinks=self.sinks,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/ops/chunked_prefill_paged_decode.py", line 356, in chunked_prefill_paged_decode
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel_paged_attention_2d[
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ](
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_ptr=output,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<37 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] USE_FP8=output_scale is not None,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 419, in
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 733, in run
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self._do_compile(key, signature, device, constexprs, options, attrs, warmup)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 861, in _do_compile
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self.compile(src, target=target, options=options.__dict__)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 300, in compile
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module = src.make_ir(target, options, codegen_fns, module_map, context)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 80, in make_ir
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return ast_to_ttir(self.fn, self, context=context, options=options, codegen_fns=codegen_fns,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module_map=module_map)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] triton.compiler.errors.CompilationError: at 106:17:
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] if USE_ALIBI_SLOPES:
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slope = tl.load(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slopes_ptr + query_head_idx, mask=head_mask, other=0.0
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_blocks = cdiv_fn(seq_len, BLOCK_SIZE)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] # iterate through tiles
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] for j in range(0, num_blocks):
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] physical_block_idx = tl.load(block_tables_ptr + block_table_offset + j)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] offs_n = tl.arange(0, BLOCK_SIZE)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] arange's range must be a power of 2
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 43, in wrapper
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return fn(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/core.py", line 1638, in arange
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return _semantic.arange(start, end)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/language/semantic.py", line 583, in arange
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise ValueError("arange's range must be a power of 2")
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ValueError: arange's range must be a power of 2
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 459, in compile_or_warm_up_model
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cuda_graph_memory_bytes = self.model_runner.capture_model()
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4586, in capture_model
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._capture_cudagraphs(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] compilation_cases=compilation_cases_decode,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] cudagraph_runtime_mode=CUDAGraphMode.FULL,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] uniform_decode=True,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4664, in _capture_cudagraphs
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self._dummy_run(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_tokens,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<6 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] activate_lora=activate_lora,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4198, in _dummy_run
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] outputs = self.model(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids=input_ids,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] **model_kwargs,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/cuda_graph.py", line 220, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.runnable(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 1231, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] hidden_states = self.model(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] input_ids, positions, intermediate_tensors, inputs_embeds
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/decorators.py", line 376, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.aot_compiled_fn(self, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_dynamo/aot_compile.py", line 124, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.fn(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/model_executor/models/qwen3_next.py", line 997, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] def forward(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/compilation/caching.py", line 54, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self.optimized_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".99", line 333, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] submod_7 = self.submod_7(getitem_21, s72, getitem_22, getitem_23, getitem_24); getitem_21 = getitem_22 = getitem_23 = submod_7 = None
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 936, in call_wrapped
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._wrapped_call(self, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 455, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] raise e
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/fx/graph_module.py", line 442, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return super(self.cls, obj).__call__(*args, **kwargs) # type: ignore[misc]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File ".107", line 5, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] unified_attention_with_output = torch.ops.vllm.unified_attention_with_output(query_8, key_8, value_9, output_5, 'model.layers.3.self_attn.attn'); query_8 = key_8 = value_9 = output_5 = unified_attention_with_output = None
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/_ops.py", line 1209, in __call__
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return self._op(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/utils/kv_transfer_utils.py", line 39, in wrapper
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/layer.py", line 923, in unified_attention_with_output
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self.impl.forward(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] self,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<7 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_block_scale=output_block_scale,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/attention/backends/rocm_attn.py", line 337, in forward
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] chunked_prefill_paged_decode(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] query=query[:num_actual_tokens],
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<17 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] sinks=self.sinks,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/attention/ops/chunked_prefill_paged_decode.py", line 356, in chunked_prefill_paged_decode
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel_paged_attention_2d[
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<3 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ](
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] output_ptr=output,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ...<37 lines>...
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] USE_FP8=output_scale is not None,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 419, in
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return lambda *args, **kwargs: self.run(grid=grid, warmup=False, *args, **kwargs)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 733, in run
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self._do_compile(key, signature, device, constexprs, options, attrs, warmup)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/runtime/jit.py", line 861, in _do_compile
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] kernel = self.compile(src, target=target, options=options.__dict__)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 300, in compile
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module = src.make_ir(target, options, codegen_fns, module_map, context)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/triton/compiler/compiler.py", line 80, in make_ir
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] return ast_to_ttir(self.fn, self, context=context, options=options, codegen_fns=codegen_fns,
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] module_map=module_map)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] triton.compiler.errors.CompilationError: at 106:17:
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] if USE_ALIBI_SLOPES:
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slope = tl.load(
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] alibi_slopes_ptr + query_head_idx, mask=head_mask, other=0.0
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] num_blocks = cdiv_fn(seq_len, BLOCK_SIZE)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] # iterate through tiles
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] for j in range(0, num_blocks):
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] physical_block_idx = tl.load(block_tables_ptr + block_table_offset + j)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] offs_n = tl.arange(0, BLOCK_SIZE)
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826] arange's range must be a power of 2
+[0;36m(Worker_TP0 pid=76495)[0;0m ERROR 12-19 17:12:51 [multiproc_executor.py:826]
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] EngineCore failed to start.
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 857, in run_engine_core
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] engine_core = EngineCoreProc(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 637, in __init__
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] super().__init__(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] vllm_config, executor_class, log_stats, executor_fail_callback
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 109, in __init__
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] num_gpu_blocks, num_cpu_blocks, kv_cache_config = self._initialize_kv_caches(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] vllm_config
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 256, in _initialize_kv_caches
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] self.model_executor.initialize_from_config(kv_cache_configs)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/abstract.py", line 116, in initialize_from_config
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] self.collective_rpc("compile_or_warm_up_model")
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 344, in get_response
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] raise RuntimeError(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] RuntimeError: Worker failed with error 'at 106:17:
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] if USE_ALIBI_SLOPES:
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] alibi_slope = tl.load(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] alibi_slopes_ptr + query_head_idx, mask=head_mask, other=0.0
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866]
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] num_blocks = cdiv_fn(seq_len, BLOCK_SIZE)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866]
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] # iterate through tiles
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] for j in range(0, num_blocks):
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] physical_block_idx = tl.load(block_tables_ptr + block_table_offset + j)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866]
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] offs_n = tl.arange(0, BLOCK_SIZE)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] ^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ERROR 12-19 17:12:51 [core.py:866] arange's range must be a power of 2', please check the stack trace above for the root cause
+[0;36m(EngineCore_DP0 pid=76413)[0;0m Process EngineCore_DP0:
+[0;36m(EngineCore_DP0 pid=76413)[0;0m Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 313, in _bootstrap
+[0;36m(EngineCore_DP0 pid=76413)[0;0m self.run()
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 108, in run
+[0;36m(EngineCore_DP0 pid=76413)[0;0m self._target(*self._args, **self._kwargs)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 870, in run_engine_core
+[0;36m(EngineCore_DP0 pid=76413)[0;0m raise e
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 857, in run_engine_core
+[0;36m(EngineCore_DP0 pid=76413)[0;0m engine_core = EngineCoreProc(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 637, in __init__
+[0;36m(EngineCore_DP0 pid=76413)[0;0m super().__init__(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m vllm_config, executor_class, log_stats, executor_fail_callback
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 109, in __init__
+[0;36m(EngineCore_DP0 pid=76413)[0;0m num_gpu_blocks, num_cpu_blocks, kv_cache_config = self._initialize_kv_caches(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m vllm_config
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 256, in _initialize_kv_caches
+[0;36m(EngineCore_DP0 pid=76413)[0;0m self.model_executor.initialize_from_config(kv_cache_configs)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/abstract.py", line 116, in initialize_from_config
+[0;36m(EngineCore_DP0 pid=76413)[0;0m self.collective_rpc("compile_or_warm_up_model")
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=76413)[0;0m return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 344, in get_response
+[0;36m(EngineCore_DP0 pid=76413)[0;0m raise RuntimeError(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=76413)[0;0m )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m RuntimeError: Worker failed with error 'at 106:17:
+[0;36m(EngineCore_DP0 pid=76413)[0;0m if USE_ALIBI_SLOPES:
+[0;36m(EngineCore_DP0 pid=76413)[0;0m alibi_slope = tl.load(
+[0;36m(EngineCore_DP0 pid=76413)[0;0m alibi_slopes_ptr + query_head_idx, mask=head_mask, other=0.0
+[0;36m(EngineCore_DP0 pid=76413)[0;0m )
+[0;36m(EngineCore_DP0 pid=76413)[0;0m
+[0;36m(EngineCore_DP0 pid=76413)[0;0m num_blocks = cdiv_fn(seq_len, BLOCK_SIZE)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m
+[0;36m(EngineCore_DP0 pid=76413)[0;0m # iterate through tiles
+[0;36m(EngineCore_DP0 pid=76413)[0;0m for j in range(0, num_blocks):
+[0;36m(EngineCore_DP0 pid=76413)[0;0m physical_block_idx = tl.load(block_tables_ptr + block_table_offset + j)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m
+[0;36m(EngineCore_DP0 pid=76413)[0;0m offs_n = tl.arange(0, BLOCK_SIZE)
+[0;36m(EngineCore_DP0 pid=76413)[0;0m ^
+[0;36m(EngineCore_DP0 pid=76413)[0;0m arange's range must be a power of 2', please check the stack trace above for the root cause
+[0;36m(Worker_TP0 pid=76495)[0;0m INFO 12-19 17:12:51 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(Worker_TP1 pid=76496)[0;0m INFO 12-19 17:12:51 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(APIServer pid=76251)[0;0m Traceback (most recent call last):
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/bin/vllm", line 7, in
+[0;36m(APIServer pid=76251)[0;0m sys.exit(main())
+[0;36m(APIServer pid=76251)[0;0m ~~~~^^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/cli/main.py", line 73, in main
+[0;36m(APIServer pid=76251)[0;0m args.dispatch_function(args)
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/cli/serve.py", line 60, in cmd
+[0;36m(APIServer pid=76251)[0;0m uvloop.run(run_server(args))
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/uvloop/__init__.py", line 96, in run
+[0;36m(APIServer pid=76251)[0;0m return __asyncio.run(
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~^
+[0;36m(APIServer pid=76251)[0;0m wrapper(),
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m ...<2 lines>...
+[0;36m(APIServer pid=76251)[0;0m **run_kwargs
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m )
+[0;36m(APIServer pid=76251)[0;0m ^
+[0;36m(APIServer pid=76251)[0;0m File "/usr/lib64/python3.13/asyncio/runners.py", line 195, in run
+[0;36m(APIServer pid=76251)[0;0m return runner.run(main)
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/usr/lib64/python3.13/asyncio/runners.py", line 118, in run
+[0;36m(APIServer pid=76251)[0;0m return self._loop.run_until_complete(task)
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/uvloop/__init__.py", line 48, in wrapper
+[0;36m(APIServer pid=76251)[0;0m return await main
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 1398, in run_server
+[0;36m(APIServer pid=76251)[0;0m await run_server_worker(listen_address, sock, args, **uvicorn_kwargs)
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 1417, in run_server_worker
+[0;36m(APIServer pid=76251)[0;0m async with build_async_engine_client(
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=76251)[0;0m args,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^
+[0;36m(APIServer pid=76251)[0;0m client_config=client_config,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m ) as engine_client:
+[0;36m(APIServer pid=76251)[0;0m ^
+[0;36m(APIServer pid=76251)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 214, in __aenter__
+[0;36m(APIServer pid=76251)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 172, in build_async_engine_client
+[0;36m(APIServer pid=76251)[0;0m async with build_async_engine_client_from_engine_args(
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=76251)[0;0m engine_args,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m ...<2 lines>...
+[0;36m(APIServer pid=76251)[0;0m client_config=client_config,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m ) as engine:
+[0;36m(APIServer pid=76251)[0;0m ^
+[0;36m(APIServer pid=76251)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 214, in __aenter__
+[0;36m(APIServer pid=76251)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 213, in build_async_engine_client_from_engine_args
+[0;36m(APIServer pid=76251)[0;0m async_llm = AsyncLLM.from_vllm_config(
+[0;36m(APIServer pid=76251)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=76251)[0;0m ...<6 lines>...
+[0;36m(APIServer pid=76251)[0;0m client_index=client_index,
+[0;36m(APIServer pid=76251)[0;0m )
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 215, in from_vllm_config
+[0;36m(APIServer pid=76251)[0;0m return cls(
+[0;36m(APIServer pid=76251)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=76251)[0;0m ...<9 lines>...
+[0;36m(APIServer pid=76251)[0;0m client_index=client_index,
+[0;36m(APIServer pid=76251)[0;0m )
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 134, in __init__
+[0;36m(APIServer pid=76251)[0;0m self.engine_core = EngineCoreClient.make_async_mp_client(
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=76251)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m ...<4 lines>...
+[0;36m(APIServer pid=76251)[0;0m client_index=client_index,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m )
+[0;36m(APIServer pid=76251)[0;0m ^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 121, in make_async_mp_client
+[0;36m(APIServer pid=76251)[0;0m return AsyncMPClient(*client_args)
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 820, in __init__
+[0;36m(APIServer pid=76251)[0;0m super().__init__(
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=76251)[0;0m asyncio_mode=True,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m ...<3 lines>...
+[0;36m(APIServer pid=76251)[0;0m client_addresses=client_addresses,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m )
+[0;36m(APIServer pid=76251)[0;0m ^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 477, in __init__
+[0;36m(APIServer pid=76251)[0;0m with launch_core_engines(vllm_config, executor_class, log_stats) as (
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 148, in __exit__
+[0;36m(APIServer pid=76251)[0;0m next(self.gen)
+[0;36m(APIServer pid=76251)[0;0m ~~~~^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/utils.py", line 903, in launch_core_engines
+[0;36m(APIServer pid=76251)[0;0m wait_for_engine_startup(
+[0;36m(APIServer pid=76251)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=76251)[0;0m handshake_socket,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m ...<5 lines>...
+[0;36m(APIServer pid=76251)[0;0m coordinator.proc if coordinator else None,
+[0;36m(APIServer pid=76251)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=76251)[0;0m )
+[0;36m(APIServer pid=76251)[0;0m ^
+[0;36m(APIServer pid=76251)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/utils.py", line 960, in wait_for_engine_startup
+[0;36m(APIServer pid=76251)[0;0m raise RuntimeError(
+[0;36m(APIServer pid=76251)[0;0m ...<3 lines>...
+[0;36m(APIServer pid=76251)[0;0m )
+[0;36m(APIServer pid=76251)[0;0m RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {}
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..3eabad4
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 14:23:22 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-af26c427-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 195.28 \nTotal input tokens: 37841 \nTotal generated tokens: 38794 \nRequest throughput (req/s): 0.92 \nOutput token throughput (tok/s): 198.66 \nPeak output token throughput (tok/s): 398.00 \nPeak concurrent requests: 14.00 \nTotal token throughput (tok/s): 392.45 \n---------------Time to First Token----------------\nMean TTFT (ms): 74.71 \nMedian TTFT (ms): 61.84 \nP99 TTFT (ms): 180.02 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 30.85 \nMedian TPOT (ms): 30.79 \nP99 TPOT (ms): 33.98 \n---------------Inter-token Latency----------------\nMean ITL (ms): 30.86 \nMedian ITL (ms): 29.96 \nP99 ITL (ms): 55.30 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..2813518
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 14:26:47 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-65595a03-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 204.73 \nTotal input tokens: 145810 \nTotal generated tokens: 152105 \nRequest throughput (req/s): 3.52 \nOutput token throughput (tok/s): 742.97 \nPeak output token throughput (tok/s): 1252.00 \nPeak concurrent requests: 55.00 \nTotal token throughput (tok/s): 1455.19 \n---------------Time to First Token----------------\nMean TTFT (ms): 81.58 \nMedian TTFT (ms): 70.11 \nP99 TTFT (ms): 184.97 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 37.28 \nMedian TPOT (ms): 36.83 \nP99 TPOT (ms): 50.83 \n---------------Inter-token Latency----------------\nMean ITL (ms): 36.98 \nMedian ITL (ms): 34.37 \nP99 ITL (ms): 119.18 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
new file mode 100644
index 0000000..3492884
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
@@ -0,0 +1,1027 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 14:22:47 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:22:47 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:22:47 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 65536, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:22:51 [model.py:514] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:22:51 [model.py:1636] Using max model len 65536
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:22:51 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:22:56 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=65536, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:22:56 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:59373 backend=nccl
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:22:56 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:22:56 [gpu_model_runner.py:3562] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:22:56 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 3.23it/s]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.08it/s]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:03<00:01, 1.16s/it]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:04<00:00, 1.24s/it]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:04<00:00, 1.12s/it]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:02 [default_loader.py:308] Loading weights took 4.48 seconds
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:02 [gpu_model_runner.py:3659] Model loading took 15.0586 GiB memory and 5.596868 seconds
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:05 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/2ba6637f7a/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:05 [backends.py:694] Dynamo bytecode transform time: 2.06 s
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:06 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:07 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 1.25 s
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:07 [monitor.py:34] torch.compile takes 3.31 s in total
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:08 [gpu_worker.py:375] Available KV cache memory: 15.33 GiB
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:09 [kv_cache_utils.py:1291] GPU KV cache size: 125,616 tokens
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:09 [kv_cache_utils.py:1296] Maximum concurrency for 65,536 tokens per request: 1.92x
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:00, 21.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 22.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:00<00:00, 23.69it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 24.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:00<00:00, 24.85it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:00<00:00, 25.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 24.62it/s]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 5.60it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 15.89it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:00<00:00, 20.87it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 24.73it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 21.00it/s]
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:11 [gpu_model_runner.py:4610] Graph capturing finished in 2 secs, took 0.94 GiB
+[0;36m(EngineCore_DP0 pid=56723)[0;0m INFO 12-19 14:23:11 [core.py:259] init engine (profile, create kv cache, warmup model) took 8.30 seconds
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=56561)[0;0m WARNING 12-19 14:23:13 [model.py:1462] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [serving_responses.py:201] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [serving_chat.py:137] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [serving_completion.py:77] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [serving_chat.py:137] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:13 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=56561)[0;0m INFO: Started server process [56561]
+[0;36m(APIServer pid=56561)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=56561)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36670 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:33 [loggers.py:248] Engine 000: Avg prompt throughput: 41.7 tokens/s, Avg generation throughput: 49.4 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:43 [loggers.py:248] Engine 000: Avg prompt throughput: 175.6 tokens/s, Avg generation throughput: 125.0 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:23:53 [loggers.py:248] Engine 000: Avg prompt throughput: 143.0 tokens/s, Avg generation throughput: 219.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:24:03 [loggers.py:248] Engine 000: Avg prompt throughput: 278.9 tokens/s, Avg generation throughput: 146.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:24:13 [loggers.py:248] Engine 000: Avg prompt throughput: 190.2 tokens/s, Avg generation throughput: 192.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:24:23 [loggers.py:248] Engine 000: Avg prompt throughput: 365.3 tokens/s, Avg generation throughput: 216.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:24:33 [loggers.py:248] Engine 000: Avg prompt throughput: 366.2 tokens/s, Avg generation throughput: 225.6 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:24:43 [loggers.py:248] Engine 000: Avg prompt throughput: 151.0 tokens/s, Avg generation throughput: 189.4 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:24:53 [loggers.py:248] Engine 000: Avg prompt throughput: 373.9 tokens/s, Avg generation throughput: 246.1 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:25:03 [loggers.py:248] Engine 000: Avg prompt throughput: 223.3 tokens/s, Avg generation throughput: 303.4 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:25:13 [loggers.py:248] Engine 000: Avg prompt throughput: 276.0 tokens/s, Avg generation throughput: 292.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:25:23 [loggers.py:248] Engine 000: Avg prompt throughput: 43.3 tokens/s, Avg generation throughput: 187.1 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:25:33 [loggers.py:248] Engine 000: Avg prompt throughput: 231.3 tokens/s, Avg generation throughput: 205.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:43744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:36204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:25:43 [loggers.py:248] Engine 000: Avg prompt throughput: 274.2 tokens/s, Avg generation throughput: 238.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:46606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:25:53 [loggers.py:248] Engine 000: Avg prompt throughput: 98.1 tokens/s, Avg generation throughput: 250.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:26:03 [loggers.py:248] Engine 000: Avg prompt throughput: 93.5 tokens/s, Avg generation throughput: 143.9 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:51166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:51182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:26:13 [loggers.py:248] Engine 000: Avg prompt throughput: 106.3 tokens/s, Avg generation throughput: 151.7 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:51166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:51166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:56432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:51166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:55530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:35880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:56442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:26:23 [loggers.py:248] Engine 000: Avg prompt throughput: 230.9 tokens/s, Avg generation throughput: 195.8 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:51166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:56442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:56448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:49106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:51182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:26:33 [loggers.py:248] Engine 000: Avg prompt throughput: 122.7 tokens/s, Avg generation throughput: 241.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:26:43 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 70.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:26:53 [loggers.py:248] Engine 000: Avg prompt throughput: 1.3 tokens/s, Avg generation throughput: 12.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:27:03 [loggers.py:248] Engine 000: Avg prompt throughput: 761.3 tokens/s, Avg generation throughput: 313.8 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 16.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:27:13 [loggers.py:248] Engine 000: Avg prompt throughput: 1216.9 tokens/s, Avg generation throughput: 732.4 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.6%, Prefix cache hit rate: 33.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:27:23 [loggers.py:248] Engine 000: Avg prompt throughput: 728.3 tokens/s, Avg generation throughput: 853.2 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.0%, Prefix cache hit rate: 40.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:27:33 [loggers.py:248] Engine 000: Avg prompt throughput: 657.9 tokens/s, Avg generation throughput: 930.2 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.8%, Prefix cache hit rate: 45.2%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:27:43 [loggers.py:248] Engine 000: Avg prompt throughput: 948.9 tokens/s, Avg generation throughput: 760.8 tokens/s, Running: 32 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.5%, Prefix cache hit rate: 44.9%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:27:53 [loggers.py:248] Engine 000: Avg prompt throughput: 922.4 tokens/s, Avg generation throughput: 928.9 tokens/s, Running: 42 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.5%, Prefix cache hit rate: 40.3%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:28:03 [loggers.py:248] Engine 000: Avg prompt throughput: 733.2 tokens/s, Avg generation throughput: 867.5 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.4%, Prefix cache hit rate: 37.2%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:28:13 [loggers.py:248] Engine 000: Avg prompt throughput: 586.3 tokens/s, Avg generation throughput: 796.1 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.4%, Prefix cache hit rate: 35.1%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:28:23 [loggers.py:248] Engine 000: Avg prompt throughput: 811.1 tokens/s, Avg generation throughput: 1103.1 tokens/s, Running: 41 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.5%, Prefix cache hit rate: 32.6%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:28:33 [loggers.py:248] Engine 000: Avg prompt throughput: 567.3 tokens/s, Avg generation throughput: 921.2 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.9%, Prefix cache hit rate: 31.0%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:28:43 [loggers.py:248] Engine 000: Avg prompt throughput: 1090.6 tokens/s, Avg generation throughput: 753.2 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.4%, Prefix cache hit rate: 28.4%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:28:53 [loggers.py:248] Engine 000: Avg prompt throughput: 605.3 tokens/s, Avg generation throughput: 650.7 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.2%, Prefix cache hit rate: 27.1%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:29:03 [loggers.py:248] Engine 000: Avg prompt throughput: 790.5 tokens/s, Avg generation throughput: 684.6 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.5%, Prefix cache hit rate: 25.6%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:46782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:29:13 [loggers.py:248] Engine 000: Avg prompt throughput: 761.7 tokens/s, Avg generation throughput: 889.6 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.5%, Prefix cache hit rate: 24.4%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:29:23 [loggers.py:248] Engine 000: Avg prompt throughput: 876.0 tokens/s, Avg generation throughput: 662.3 tokens/s, Running: 29 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.2%, Prefix cache hit rate: 23.1%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:46782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:46782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:29:33 [loggers.py:248] Engine 000: Avg prompt throughput: 483.5 tokens/s, Avg generation throughput: 766.6 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 22.4%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:46782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:29:43 [loggers.py:248] Engine 000: Avg prompt throughput: 719.3 tokens/s, Avg generation throughput: 763.7 tokens/s, Running: 34 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.6%, Prefix cache hit rate: 21.4%
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:48676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:41448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:53154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:60482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:54650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO: 127.0.0.1:45082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:29:53 [loggers.py:248] Engine 000: Avg prompt throughput: 1274.6 tokens/s, Avg generation throughput: 871.3 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.1%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:30:03 [loggers.py:248] Engine 000: Avg prompt throughput: 45.2 tokens/s, Avg generation throughput: 699.4 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.4%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:30:13 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 222.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=56561)[0;0m INFO 12-19 14:30:18 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1219 14:30:18.025563119 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
new file mode 100644
index 0000000..27d8d51
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 389.5727865589979,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 2.5669144111239337,
+ "tokens_per_second": 1890.096088342886
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..977e945
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 15:41:39 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-819f7ea7-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 189.28 \nTotal input tokens: 37841 \nTotal generated tokens: 38766 \nRequest throughput (req/s): 0.95 \nOutput token throughput (tok/s): 204.80 \nPeak output token throughput (tok/s): 415.00 \nPeak concurrent requests: 10.00 \nTotal token throughput (tok/s): 404.72 \n---------------Time to First Token----------------\nMean TTFT (ms): 80.83 \nMedian TTFT (ms): 50.95 \nP99 TTFT (ms): 232.06 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 19.52 \nMedian TPOT (ms): 19.18 \nP99 TPOT (ms): 24.48 \n---------------Inter-token Latency----------------\nMean ITL (ms): 19.42 \nMedian ITL (ms): 18.84 \nP99 ITL (ms): 34.36 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..ef40343
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 15:44:56 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-9c7b74a1-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 196.04 \nTotal input tokens: 145810 \nTotal generated tokens: 152194 \nRequest throughput (req/s): 3.67 \nOutput token throughput (tok/s): 776.36 \nPeak output token throughput (tok/s): 1303.00 \nPeak concurrent requests: 46.00 \nTotal token throughput (tok/s): 1520.15 \n---------------Time to First Token----------------\nMean TTFT (ms): 86.17 \nMedian TTFT (ms): 57.78 \nP99 TTFT (ms): 291.16 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 28.21 \nMedian TPOT (ms): 27.26 \nP99 TPOT (ms): 51.74 \n---------------Inter-token Latency----------------\nMean ITL (ms): 27.56 \nMedian ITL (ms): 23.38 \nP99 ITL (ms): 140.32 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log
new file mode 100644
index 0000000..1849592
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log
@@ -0,0 +1,1034 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 15:40:52 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:40:52 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:40:52 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 65536, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:40:56 [model.py:514] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:40:56 [model.py:1636] Using max model len 65536
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:40:57 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=62228)[0;0m INFO 12-19 15:41:01 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=65536, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=62228)[0;0m WARNING 12-19 15:41:01 [multiproc_executor.py:884] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+INFO 12-19 15:41:04 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:40771 backend=nccl
+INFO 12-19 15:41:04 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:40771 backend=nccl
+INFO 12-19 15:41:04 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-19 15:41:05 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-19 15:41:05 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:05 [gpu_model_runner.py:3562] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(Worker_TP1 pid=62311)[0;0m INFO 12-19 15:41:06 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:06 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=62310)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 7.96it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.60it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.27it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.24it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.37it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:10 [default_loader.py:308] Loading weights took 2.92 seconds
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:11 [gpu_model_runner.py:3659] Model loading took 7.5820 GiB memory and 4.299710 seconds
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:13 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/2936335609/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:13 [backends.py:694] Dynamo bytecode transform time: 2.19 s
+[0;36m(Worker_TP1 pid=62311)[0;0m INFO 12-19 15:41:14 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:14 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:16 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 1.36 s
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:16 [monitor.py:34] torch.compile takes 3.55 s in total
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:17 [gpu_worker.py:375] Available KV cache memory: 22.92 GiB
+[0;36m(EngineCore_DP0 pid=62228)[0;0m INFO 12-19 15:41:18 [kv_cache_utils.py:1291] GPU KV cache size: 375,536 tokens
+[0;36m(EngineCore_DP0 pid=62228)[0;0m INFO 12-19 15:41:18 [kv_cache_utils.py:1296] Maximum concurrency for 65,536 tokens per request: 5.73x
+[0;36m(Worker_TP0 pid=62310)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:05, 3.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:04, 3.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:04, 3.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:01<00:04, 3.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:01<00:03, 3.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:01<00:03, 3.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:01<00:03, 3.61it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:02<00:03, 3.62it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:02<00:02, 3.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:02<00:02, 3.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:03<00:02, 3.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:03<00:01, 3.68it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:03<00:01, 3.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:03<00:01, 3.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:04<00:01, 3.75it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:04<00:00, 3.74it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:04<00:00, 3.76it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:04<00:00, 3.76it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:05<00:00, 3.78it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:05<00:00, 3.68it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:04, 2.37it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:02, 3.05it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:02, 3.38it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:01<00:01, 3.55it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:01<00:01, 3.67it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:01<00:01, 3.76it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:01<00:01, 3.80it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:02<00:00, 3.82it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:02<00:00, 3.85it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:02<00:00, 3.84it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:03<00:00, 3.85it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:03<00:00, 3.66it/s]
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:41:26 [gpu_model_runner.py:4610] Graph capturing finished in 9 secs, took 0.62 GiB
+[0;36m(EngineCore_DP0 pid=62228)[0;0m INFO 12-19 15:41:26 [core.py:259] init engine (profile, create kv cache, warmup model) took 15.91 seconds
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:28 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=62066)[0;0m WARNING 12-19 15:41:28 [model.py:1462] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:28 [serving_responses.py:201] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:28 [serving_chat.py:137] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [serving_completion.py:77] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [serving_chat.py:137] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:29 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=62066)[0;0m INFO: Started server process [62066]
+[0;36m(APIServer pid=62066)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=62066)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:36254 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:49 [loggers.py:248] Engine 000: Avg prompt throughput: 41.7 tokens/s, Avg generation throughput: 66.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:41:59 [loggers.py:248] Engine 000: Avg prompt throughput: 175.6 tokens/s, Avg generation throughput: 160.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:42:09 [loggers.py:248] Engine 000: Avg prompt throughput: 143.0 tokens/s, Avg generation throughput: 206.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:42:19 [loggers.py:248] Engine 000: Avg prompt throughput: 278.9 tokens/s, Avg generation throughput: 142.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:32896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:42:29 [loggers.py:248] Engine 000: Avg prompt throughput: 220.0 tokens/s, Avg generation throughput: 186.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:42:39 [loggers.py:248] Engine 000: Avg prompt throughput: 335.5 tokens/s, Avg generation throughput: 239.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:35678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:42:49 [loggers.py:248] Engine 000: Avg prompt throughput: 366.2 tokens/s, Avg generation throughput: 227.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:42:59 [loggers.py:248] Engine 000: Avg prompt throughput: 151.0 tokens/s, Avg generation throughput: 171.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:43:09 [loggers.py:248] Engine 000: Avg prompt throughput: 373.9 tokens/s, Avg generation throughput: 272.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:43:19 [loggers.py:248] Engine 000: Avg prompt throughput: 223.3 tokens/s, Avg generation throughput: 353.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:59988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:43:29 [loggers.py:248] Engine 000: Avg prompt throughput: 276.0 tokens/s, Avg generation throughput: 204.4 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:43:39 [loggers.py:248] Engine 000: Avg prompt throughput: 43.3 tokens/s, Avg generation throughput: 199.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:43:49 [loggers.py:248] Engine 000: Avg prompt throughput: 231.3 tokens/s, Avg generation throughput: 197.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:43:59 [loggers.py:248] Engine 000: Avg prompt throughput: 277.9 tokens/s, Avg generation throughput: 255.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:58222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:44:09 [loggers.py:248] Engine 000: Avg prompt throughput: 94.4 tokens/s, Avg generation throughput: 274.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:36756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:44:19 [loggers.py:248] Engine 000: Avg prompt throughput: 124.0 tokens/s, Avg generation throughput: 67.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:36756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:44:29 [loggers.py:248] Engine 000: Avg prompt throughput: 75.8 tokens/s, Avg generation throughput: 198.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:36756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:36756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:44:39 [loggers.py:248] Engine 000: Avg prompt throughput: 230.9 tokens/s, Avg generation throughput: 196.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:57768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:57784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:36756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:44:49 [loggers.py:248] Engine 000: Avg prompt throughput: 122.7 tokens/s, Avg generation throughput: 238.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:44:59 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 29.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:45:09 [loggers.py:248] Engine 000: Avg prompt throughput: 439.7 tokens/s, Avg generation throughput: 263.7 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 9.9%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:45:19 [loggers.py:248] Engine 000: Avg prompt throughput: 1232.0 tokens/s, Avg generation throughput: 770.6 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 29.4%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:45:29 [loggers.py:248] Engine 000: Avg prompt throughput: 926.1 tokens/s, Avg generation throughput: 840.9 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 39.1%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:45:39 [loggers.py:248] Engine 000: Avg prompt throughput: 634.2 tokens/s, Avg generation throughput: 971.0 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 44.2%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:45:49 [loggers.py:248] Engine 000: Avg prompt throughput: 720.9 tokens/s, Avg generation throughput: 731.9 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 47.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:45:59 [loggers.py:248] Engine 000: Avg prompt throughput: 914.1 tokens/s, Avg generation throughput: 897.0 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 42.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:46:09 [loggers.py:248] Engine 000: Avg prompt throughput: 890.5 tokens/s, Avg generation throughput: 892.2 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 38.1%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:46:19 [loggers.py:248] Engine 000: Avg prompt throughput: 469.5 tokens/s, Avg generation throughput: 790.4 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 36.3%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:46:29 [loggers.py:248] Engine 000: Avg prompt throughput: 1005.2 tokens/s, Avg generation throughput: 1015.6 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 33.0%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:46:39 [loggers.py:248] Engine 000: Avg prompt throughput: 527.3 tokens/s, Avg generation throughput: 1004.4 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 31.5%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:46:49 [loggers.py:248] Engine 000: Avg prompt throughput: 948.7 tokens/s, Avg generation throughput: 764.8 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 29.7%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:46:59 [loggers.py:248] Engine 000: Avg prompt throughput: 907.5 tokens/s, Avg generation throughput: 628.2 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 27.7%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:47:09 [loggers.py:248] Engine 000: Avg prompt throughput: 529.9 tokens/s, Avg generation throughput: 681.8 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 26.6%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:47:19 [loggers.py:248] Engine 000: Avg prompt throughput: 795.3 tokens/s, Avg generation throughput: 889.5 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 25.3%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:47:29 [loggers.py:248] Engine 000: Avg prompt throughput: 876.4 tokens/s, Avg generation throughput: 697.8 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 23.9%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:47:39 [loggers.py:248] Engine 000: Avg prompt throughput: 654.3 tokens/s, Avg generation throughput: 733.1 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 22.9%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:47:49 [loggers.py:248] Engine 000: Avg prompt throughput: 595.1 tokens/s, Avg generation throughput: 764.6 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 22.1%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:37236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:56314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:47:59 [loggers.py:248] Engine 000: Avg prompt throughput: 1197.2 tokens/s, Avg generation throughput: 803.1 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:33548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:46310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:44256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:44272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:60042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:50072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:45446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO: 127.0.0.1:40296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:48:09 [loggers.py:248] Engine 000: Avg prompt throughput: 317.8 tokens/s, Avg generation throughput: 867.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 20.3%
+[0;36m(APIServer pid=62066)[0;0m INFO 12-19 15:48:18 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP1 pid=62311)[0;0m INFO 12-19 15:48:18 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(Worker_TP0 pid=62310)[0;0m INFO 12-19 15:48:18 [multiproc_executor.py:711] Parent process exited, terminating worker
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..d7c2b30
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 14:42:05 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-3e31d4e4-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 215.02 \nTotal input tokens: 38756 \nTotal generated tokens: 39194 \nRequest throughput (req/s): 0.84 \nOutput token throughput (tok/s): 182.28 \nPeak output token throughput (tok/s): 324.00 \nPeak concurrent requests: 19.00 \nTotal token throughput (tok/s): 362.52 \n---------------Time to First Token----------------\nMean TTFT (ms): 118.15 \nMedian TTFT (ms): 106.93 \nP99 TTFT (ms): 227.27 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 54.56 \nMedian TPOT (ms): 55.08 \nP99 TPOT (ms): 69.46 \n---------------Inter-token Latency----------------\nMean ITL (ms): 54.82 \nMedian ITL (ms): 54.23 \nP99 ITL (ms): 120.82 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..6192364
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "WARNING 12-19 14:45:51 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.\nNamespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-84255adf-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 238.83 \nTotal input tokens: 145540 \nTotal generated tokens: 151955 \nRequest throughput (req/s): 3.01 \nOutput token throughput (tok/s): 636.24 \nPeak output token throughput (tok/s): 960.00 \nPeak concurrent requests: 94.00 \nTotal token throughput (tok/s): 1245.62 \n---------------Time to First Token----------------\nMean TTFT (ms): 1170.97 \nMedian TTFT (ms): 182.43 \nP99 TTFT (ms): 5964.07 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 73.13 \nMedian TPOT (ms): 74.02 \nP99 TPOT (ms): 87.28 \n---------------Inter-token Latency----------------\nMean ITL (ms): 72.97 \nMedian ITL (ms): 68.87 \nP99 ITL (ms): 158.23 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_server.log
new file mode 100644
index 0000000..da9466a
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_server.log
@@ -0,0 +1,1033 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 14:41:24 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:24 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:24 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=57644)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:28 [model.py:514] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=57644)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 6.78it/s]
Parse safetensors files: 67%|██████▋ | 2/3 [00:00<00:00, 4.85it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 7.59it/s]
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:29 [model.py:1636] Using max model len 32768
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:29 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:29 [config.py:271] Overriding max cuda graph capture size to 1024 for performance.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:33 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:33 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:59097 backend=nccl
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:33 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:34 [gpu_model_runner.py:3562] Starting to load model openai/gpt-oss-20b...
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:34 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:34 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:34 [mxfp4.py:170] Using Triton backend
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:01, 1.08it/s]
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:02<00:01, 1.15s/it]
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:03<00:00, 1.22s/it]
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:03<00:00, 1.18s/it]
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:38 [default_loader.py:308] Loading weights took 3.63 seconds
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:39 [gpu_model_runner.py:3659] Model loading took 14.3066 GiB memory and 4.313271 seconds
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:41 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/4c3c97b6bf/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:41 [backends.py:694] Dynamo bytecode transform time: 1.69 s
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:42 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:43 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 1.18 s
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:43 [monitor.py:34] torch.compile takes 2.87 s in total
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:44 [gpu_worker.py:375] Available KV cache memory: 15.77 GiB
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:44 [kv_cache_utils.py:1291] GPU KV cache size: 344,560 tokens
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:44 [kv_cache_utils.py:1296] Maximum concurrency for 32,768 tokens per request: 19.71x
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 1%| | 1/83 [00:00<00:14, 5.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:13, 6.12it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 4%|▎ | 3/83 [00:00<00:12, 6.21it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:00<00:12, 6.38it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 6%|▌ | 5/83 [00:00<00:12, 6.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:00<00:11, 6.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 8%|▊ | 7/83 [00:01<00:11, 6.61it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:01<00:11, 6.69it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 9/83 [00:01<00:10, 6.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:01<00:10, 6.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 13%|█▎ | 11/83 [00:01<00:10, 6.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:01<00:10, 7.04it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 13/83 [00:01<00:09, 7.12it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:02<00:09, 7.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 15/83 [00:02<00:09, 7.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:02<00:09, 7.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 20%|██ | 17/83 [00:02<00:08, 7.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:02<00:08, 7.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 23%|██▎ | 19/83 [00:02<00:08, 7.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:02<00:08, 7.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 25%|██▌ | 21/83 [00:02<00:07, 7.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:03<00:07, 7.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 28%|██▊ | 23/83 [00:03<00:07, 7.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:03<00:07, 8.12it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 30%|███ | 25/83 [00:03<00:07, 8.25it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:03<00:06, 8.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 33%|███▎ | 27/83 [00:03<00:06, 8.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:03<00:06, 8.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 35%|███▍ | 29/83 [00:03<00:06, 8.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 31/83 [00:04<00:05, 9.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 40%|███▉ | 33/83 [00:04<00:05, 9.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 35/83 [00:04<00:04, 10.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▍ | 37/83 [00:04<00:04, 10.33it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 39/83 [00:04<00:04, 10.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 49%|████▉ | 41/83 [00:05<00:03, 10.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 52%|█████▏ | 43/83 [00:05<00:03, 11.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 54%|█████▍ | 45/83 [00:05<00:03, 11.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 57%|█████▋ | 47/83 [00:05<00:02, 12.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 59%|█████▉ | 49/83 [00:05<00:02, 13.15it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 61%|██████▏ | 51/83 [00:05<00:02, 13.79it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▍ | 53/83 [00:05<00:02, 14.34it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 66%|██████▋ | 55/83 [00:05<00:01, 14.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 69%|██████▊ | 57/83 [00:06<00:01, 15.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 71%|███████ | 59/83 [00:06<00:01, 15.76it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 61/83 [00:06<00:01, 16.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 76%|███████▌ | 63/83 [00:06<00:01, 16.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 78%|███████▊ | 65/83 [00:06<00:01, 17.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:06<00:00, 18.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 86%|████████▌ | 71/83 [00:06<00:00, 19.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:06<00:00, 21.19it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 93%|█████████▎| 77/83 [00:07<00:00, 22.40it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 96%|█████████▋| 80/83 [00:07<00:00, 23.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:07<00:00, 19.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:07<00:00, 11.22it/s]
+[0;36m(EngineCore_DP0 pid=57811)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 26.05it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 28.11it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 25.77it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 23.53it/s]
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:53 [gpu_model_runner.py:4610] Graph capturing finished in 9 secs, took 1.41 GiB
+[0;36m(EngineCore_DP0 pid=57811)[0;0m INFO 12-19 14:41:53 [core.py:259] init engine (profile, create kv cache, warmup model) took 14.23 seconds
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:54 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=57644)[0;0m WARNING 12-19 14:41:54 [serving_responses.py:222] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:41:55 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=57644)[0;0m INFO: Started server process [57644]
+[0;36m(APIServer pid=57644)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=57644)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:44822 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:42:15 [loggers.py:248] Engine 000: Avg prompt throughput: 4.9 tokens/s, Avg generation throughput: 15.3 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:42:25 [loggers.py:248] Engine 000: Avg prompt throughput: 132.7 tokens/s, Avg generation throughput: 85.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:42:35 [loggers.py:248] Engine 000: Avg prompt throughput: 223.5 tokens/s, Avg generation throughput: 149.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:42:45 [loggers.py:248] Engine 000: Avg prompt throughput: 275.0 tokens/s, Avg generation throughput: 176.1 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:42:55 [loggers.py:248] Engine 000: Avg prompt throughput: 223.7 tokens/s, Avg generation throughput: 183.3 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:43:05 [loggers.py:248] Engine 000: Avg prompt throughput: 366.7 tokens/s, Avg generation throughput: 194.7 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:52022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:52036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:43:15 [loggers.py:248] Engine 000: Avg prompt throughput: 331.0 tokens/s, Avg generation throughput: 218.7 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:45276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:43:25 [loggers.py:248] Engine 000: Avg prompt throughput: 275.3 tokens/s, Avg generation throughput: 213.6 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:52036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:52022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:43:35 [loggers.py:248] Engine 000: Avg prompt throughput: 288.6 tokens/s, Avg generation throughput: 199.0 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:43:45 [loggers.py:248] Engine 000: Avg prompt throughput: 174.4 tokens/s, Avg generation throughput: 257.5 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:43:55 [loggers.py:248] Engine 000: Avg prompt throughput: 377.1 tokens/s, Avg generation throughput: 276.8 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:44:05 [loggers.py:248] Engine 000: Avg prompt throughput: 42.4 tokens/s, Avg generation throughput: 291.9 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:52022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:44:15 [loggers.py:248] Engine 000: Avg prompt throughput: 196.8 tokens/s, Avg generation throughput: 188.3 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:57980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:44:25 [loggers.py:248] Engine 000: Avg prompt throughput: 242.5 tokens/s, Avg generation throughput: 248.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:52022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:44:35 [loggers.py:248] Engine 000: Avg prompt throughput: 156.5 tokens/s, Avg generation throughput: 217.1 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:44:45 [loggers.py:248] Engine 000: Avg prompt throughput: 63.4 tokens/s, Avg generation throughput: 182.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:59044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:60248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:44:55 [loggers.py:248] Engine 000: Avg prompt throughput: 156.7 tokens/s, Avg generation throughput: 165.4 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:45:05 [loggers.py:248] Engine 000: Avg prompt throughput: 89.6 tokens/s, Avg generation throughput: 185.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:36256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:40676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:53404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:45:15 [loggers.py:248] Engine 000: Avg prompt throughput: 256.0 tokens/s, Avg generation throughput: 221.2 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:45:25 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 184.1 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:45:35 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 43.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:45:45 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 26.3 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:45:55 [loggers.py:248] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 9.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:46:05 [loggers.py:248] Engine 000: Avg prompt throughput: 359.9 tokens/s, Avg generation throughput: 112.5 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 8.1%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:46:15 [loggers.py:248] Engine 000: Avg prompt throughput: 1197.2 tokens/s, Avg generation throughput: 388.8 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.0%, Prefix cache hit rate: 27.6%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:46:25 [loggers.py:248] Engine 000: Avg prompt throughput: 1114.5 tokens/s, Avg generation throughput: 664.9 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 39.3%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:46:35 [loggers.py:248] Engine 000: Avg prompt throughput: 638.2 tokens/s, Avg generation throughput: 721.7 tokens/s, Running: 59 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 44.3%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:46:45 [loggers.py:248] Engine 000: Avg prompt throughput: 582.4 tokens/s, Avg generation throughput: 815.7 tokens/s, Running: 58 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 47.9%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:46:55 [loggers.py:248] Engine 000: Avg prompt throughput: 983.1 tokens/s, Avg generation throughput: 787.0 tokens/s, Running: 60 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 42.5%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:47:05 [loggers.py:248] Engine 000: Avg prompt throughput: 694.6 tokens/s, Avg generation throughput: 820.6 tokens/s, Running: 64 reqs, Waiting: 11 reqs, GPU KV cache usage: 4.9%, Prefix cache hit rate: 39.4%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:47:15 [loggers.py:248] Engine 000: Avg prompt throughput: 739.8 tokens/s, Avg generation throughput: 848.4 tokens/s, Running: 63 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 36.5%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:47:25 [loggers.py:248] Engine 000: Avg prompt throughput: 446.1 tokens/s, Avg generation throughput: 887.8 tokens/s, Running: 64 reqs, Waiting: 27 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 35.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:39056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:47:35 [loggers.py:248] Engine 000: Avg prompt throughput: 714.1 tokens/s, Avg generation throughput: 876.7 tokens/s, Running: 64 reqs, Waiting: 15 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 32.8%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:39056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:47:45 [loggers.py:248] Engine 000: Avg prompt throughput: 873.1 tokens/s, Avg generation throughput: 851.2 tokens/s, Running: 64 reqs, Waiting: 13 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 30.4%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:47:55 [loggers.py:248] Engine 000: Avg prompt throughput: 1214.1 tokens/s, Avg generation throughput: 789.9 tokens/s, Running: 57 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 28.2%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:48:05 [loggers.py:248] Engine 000: Avg prompt throughput: 573.6 tokens/s, Avg generation throughput: 800.4 tokens/s, Running: 57 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 27.1%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:39056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:48:15 [loggers.py:248] Engine 000: Avg prompt throughput: 682.6 tokens/s, Avg generation throughput: 835.3 tokens/s, Running: 63 reqs, Waiting: 8 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 26.0%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:48:25 [loggers.py:248] Engine 000: Avg prompt throughput: 1010.3 tokens/s, Avg generation throughput: 764.0 tokens/s, Running: 47 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 24.3%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:39056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:48:35 [loggers.py:248] Engine 000: Avg prompt throughput: 773.4 tokens/s, Avg generation throughput: 738.9 tokens/s, Running: 63 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 23.2%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:48:45 [loggers.py:248] Engine 000: Avg prompt throughput: 455.7 tokens/s, Avg generation throughput: 789.5 tokens/s, Running: 58 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 22.5%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:41550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:39056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:48:55 [loggers.py:248] Engine 000: Avg prompt throughput: 1172.2 tokens/s, Avg generation throughput: 780.2 tokens/s, Running: 63 reqs, Waiting: 1 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 21.1%
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:47946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:49608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:56924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:55108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:49614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:49620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:46028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:42604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:43450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO: 127.0.0.1:51670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:49:05 [loggers.py:248] Engine 000: Avg prompt throughput: 328.7 tokens/s, Avg generation throughput: 876.7 tokens/s, Running: 57 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:49:15 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 587.6 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:49:25 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 278.1 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:49:35 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 126.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:49:45 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 31.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:49:55 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 24.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=57644)[0;0m INFO 12-19 14:49:58 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1219 14:49:58.849348158 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_throughput.json
new file mode 100644
index 0000000..0f4fe9b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 604.6067389169984,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 1.6539676712688476,
+ "tokens_per_second": 1221.9380837920544
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp2_server.log b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp2_server.log
new file mode 100644
index 0000000..2cb7082
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-rocm_atten/openai_gpt-oss-20b_tp2_server.log
@@ -0,0 +1,437 @@
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+WARNING 12-19 17:07:51 [attention.py:82] Using VLLM_V1_USE_PREFILL_DECODE_ATTENTION environment variable is deprecated and will be removed in v0.14.0 or v1.0.0, whichever is soonest. Please use --attention-config.use_prefill_decode_attention command line argument or AttentionConfig(use_prefill_decode_attention=...) config field instead.
+[0;36m(APIServer pid=74952)[0;0m INFO 12-19 17:07:51 [api_server.py:1351] vLLM API server version 0.13.0rc2.dev112+g763963aa7.d20251213
+[0;36m(APIServer pid=74952)[0;0m INFO 12-19 17:07:51 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 32768, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=74952)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=74952)[0;0m INFO 12-19 17:07:55 [model.py:514] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=74952)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 6.81it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 20.38it/s]
+[0;36m(APIServer pid=74952)[0;0m INFO 12-19 17:07:55 [model.py:1636] Using max model len 32768
+[0;36m(APIServer pid=74952)[0;0m INFO 12-19 17:07:56 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=74952)[0;0m INFO 12-19 17:07:56 [config.py:271] Overriding max cuda graph capture size to 1024 for performance.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+[0;36m(EngineCore_DP0 pid=75118)[0;0m INFO 12-19 17:08:00 [core.py:93] Initializing a V1 LLM engine (v0.13.0rc2.dev112+g763963aa7.d20251213) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=75118)[0;0m WARNING 12-19 17:08:00 [multiproc_executor.py:884] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+Skipping import of cpp extensions due to incompatible torch version 2.10.0a0+rocm7.11.0a20251210 for torchao version 0.14.1 Please see https://github.com/pytorch/ao/issues/2919 for more info
+INFO 12-19 17:08:03 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:58633 backend=nccl
+INFO 12-19 17:08:03 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:58633 backend=nccl
+INFO 12-19 17:08:04 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-19 17:08:04 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+INFO 12-19 17:08:04 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:05 [gpu_model_runner.py:3562] Starting to load model openai/gpt-oss-20b...
+[0;36m(Worker_TP1 pid=75201)[0;0m INFO 12-19 17:08:05 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=75201)[0;0m INFO 12-19 17:08:05 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP1 pid=75201)[0;0m INFO 12-19 17:08:05 [mxfp4.py:170] Using Triton backend
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:05 [rocm.py:306] Using Rocm Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:05 [layer.py:372] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:05 [mxfp4.py:170] Using Triton backend
+[0;36m(Worker_TP0 pid=75200)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=75200)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:01, 1.12it/s]
+[0;36m(Worker_TP0 pid=75200)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:01<00:00, 1.12it/s]
+[0;36m(Worker_TP0 pid=75200)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.14it/s]
+[0;36m(Worker_TP0 pid=75200)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.14it/s]
+[0;36m(Worker_TP0 pid=75200)[0;0m
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:08 [default_loader.py:308] Loading weights took 2.68 seconds
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:09 [gpu_model_runner.py:3659] Model loading took 7.4551 GiB memory and 3.520081 seconds
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:11 [backends.py:634] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/a6064e95d6/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:11 [backends.py:694] Dynamo bytecode transform time: 1.83 s
+[0;36m(Worker_TP1 pid=75201)[0;0m INFO 12-19 17:08:12 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:12 [backends.py:261] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:13 [backends.py:278] Compiling a graph for compile range (1, 2048) takes 1.26 s
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:13 [monitor.py:34] torch.compile takes 3.09 s in total
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:15 [gpu_worker.py:375] Available KV cache memory: 22.75 GiB
+[0;36m(EngineCore_DP0 pid=75118)[0;0m INFO 12-19 17:08:15 [kv_cache_utils.py:1291] GPU KV cache size: 993,760 tokens
+[0;36m(EngineCore_DP0 pid=75118)[0;0m INFO 12-19 17:08:15 [kv_cache_utils.py:1296] Maximum concurrency for 32,768 tokens per request: 56.85x
+[0;36m(Worker_TP0 pid=75200)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 1%| | 1/83 [00:00<00:30, 2.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:29, 2.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 4%|▎ | 3/83 [00:01<00:29, 2.74it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:01<00:28, 2.75it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 6%|▌ | 5/83 [00:01<00:28, 2.77it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:02<00:27, 2.78it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 8%|▊ | 7/83 [00:02<00:27, 2.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:02<00:26, 2.82it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 9/83 [00:03<00:26, 2.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:03<00:25, 2.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 13%|█▎ | 11/83 [00:03<00:25, 2.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:04<00:24, 2.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 13/83 [00:04<00:23, 2.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:04<00:23, 2.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 15/83 [00:05<00:23, 2.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:05<00:22, 2.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 20%|██ | 17/83 [00:05<00:22, 2.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:06<00:21, 2.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 23%|██▎ | 19/83 [00:06<00:21, 2.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:06<00:21, 2.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 25%|██▌ | 21/83 [00:07<00:20, 2.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:07<00:20, 3.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 28%|██▊ | 23/83 [00:07<00:19, 3.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:08<00:19, 3.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 30%|███ | 25/83 [00:08<00:18, 3.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:08<00:18, 3.12it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 33%|███▎ | 27/83 [00:09<00:17, 3.13it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:09<00:17, 3.16it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 35%|███▍ | 29/83 [00:09<00:16, 3.19it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▌ | 30/83 [00:10<00:16, 3.22it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 31/83 [00:10<00:16, 3.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 39%|███▊ | 32/83 [00:10<00:15, 3.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 40%|███▉ | 33/83 [00:11<00:15, 3.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 41%|████ | 34/83 [00:11<00:14, 3.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 35/83 [00:11<00:14, 3.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 43%|████▎ | 36/83 [00:11<00:14, 3.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▍ | 37/83 [00:12<00:13, 3.33it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 46%|████▌ | 38/83 [00:12<00:13, 3.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 39/83 [00:12<00:13, 3.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 48%|████▊ | 40/83 [00:13<00:12, 3.40it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 49%|████▉ | 41/83 [00:13<00:12, 3.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 51%|█████ | 42/83 [00:13<00:11, 3.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 52%|█████▏ | 43/83 [00:13<00:11, 3.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 44/83 [00:14<00:11, 3.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 54%|█████▍ | 45/83 [00:14<00:10, 3.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▌ | 46/83 [00:14<00:10, 3.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 57%|█████▋ | 47/83 [00:15<00:10, 3.58it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 48/83 [00:15<00:09, 3.62it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 59%|█████▉ | 49/83 [00:15<00:09, 3.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 60%|██████ | 50/83 [00:15<00:09, 3.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 61%|██████▏ | 51/83 [00:16<00:08, 3.70it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 52/83 [00:16<00:08, 3.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▍ | 53/83 [00:16<00:08, 3.75it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 65%|██████▌ | 54/83 [00:16<00:07, 3.75it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 66%|██████▋ | 55/83 [00:17<00:07, 3.76it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 67%|██████▋ | 56/83 [00:17<00:07, 3.77it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 69%|██████▊ | 57/83 [00:17<00:06, 3.79it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 70%|██████▉ | 58/83 [00:17<00:06, 3.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 71%|███████ | 59/83 [00:18<00:06, 3.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 72%|███████▏ | 60/83 [00:18<00:05, 3.87it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 61/83 [00:18<00:05, 3.90it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 75%|███████▍ | 62/83 [00:18<00:05, 3.90it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 76%|███████▌ | 63/83 [00:19<00:05, 3.90it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 77%|███████▋ | 64/83 [00:19<00:04, 3.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 78%|███████▊ | 65/83 [00:19<00:04, 3.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 80%|███████▉ | 66/83 [00:20<00:04, 3.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 81%|████████ | 67/83 [00:20<00:04, 3.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:20<00:03, 3.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 83%|████████▎ | 69/83 [00:20<00:03, 3.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 70/83 [00:21<00:03, 3.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 86%|████████▌ | 71/83 [00:21<00:03, 3.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 87%|████████▋ | 72/83 [00:21<00:02, 4.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 88%|████████▊ | 73/83 [00:21<00:02, 4.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:21<00:02, 4.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 90%|█████████ | 75/83 [00:22<00:01, 4.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 92%|█████████▏| 76/83 [00:22<00:01, 4.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 93%|█████████▎| 77/83 [00:22<00:01, 4.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 94%|█████████▍| 78/83 [00:22<00:01, 4.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▌| 79/83 [00:23<00:00, 4.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 96%|█████████▋| 80/83 [00:23<00:00, 4.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 98%|█████████▊| 81/83 [00:23<00:00, 3.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 99%|█████████▉| 82/83 [00:24<00:00, 3.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:24<00:00, 3.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:24<00:00, 3.42it/s]
+[0;36m(Worker_TP0 pid=75200)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:02, 4.16it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:02, 4.39it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:01, 4.45it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:01, 4.45it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:01<00:01, 4.44it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:01<00:01, 4.44it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:01<00:00, 4.44it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:01<00:00, 4.44it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:02<00:00, 4.41it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:02<00:00, 4.40it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:02<00:00, 4.35it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:02<00:00, 4.40it/s]
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:43 [gpu_model_runner.py:4610] Graph capturing finished in 27 secs, took 0.92 GiB
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] WorkerProc hit an exception.
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 150.00 MiB. GPU 0 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.76 GiB is allocated by PyTorch, with 106.00 MiB allocated in private pools (e.g., HIP Graphs), and 135.04 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] WorkerProc hit an exception.
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 150.00 MiB. GPU 1 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.76 GiB is allocated by PyTorch, with 106.00 MiB allocated in private pools (e.g., HIP Graphs), and 119.04 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4302, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 150.00 MiB. GPU 0 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.76 GiB is allocated by PyTorch, with 106.00 MiB allocated in private pools (e.g., HIP Graphs), and 135.04 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP0 pid=75200)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampler_output = self.sampler(
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits=logits, sampling_metadata=dummy_metadata
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 96, in forward
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampled, processed_logprobs = self.sample(logits, sampling_metadata)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/sampler.py", line 187, in sample
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] random_sampled, processed_logprobs = self.topk_topp_sampler(
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits,
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<2 lines>...
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] sampling_metadata.top_p,
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] )
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1776, in _wrapped_call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return self._call_impl(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/nn/modules/module.py", line 1787, in _call_impl
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return forward_call(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 104, in forward_native
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits = self.apply_top_k_top_p(logits, k, p)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/sample/ops/topk_topp_sampler.py", line 258, in apply_top_k_top_p
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] logits_sort, logits_idx = logits.sort(dim=-1, descending=False)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] torch.OutOfMemoryError: HIP out of memory. Tried to allocate 150.00 MiB. GPU 1 has a total capacity of 31.86 GiB of which 0 bytes is free. Of the allocated memory 30.76 GiB is allocated by PyTorch, with 106.00 MiB allocated in private pools (e.g., HIP Graphs), and 119.04 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] The above exception was the direct cause of the following exception:
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] Traceback (most recent call last):
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 821, in worker_busy_loop
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] output = func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_worker.py", line 538, in compile_or_warm_up_model
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] self.model_runner._dummy_sampler_run(hidden_states=last_hidden_states)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/torch/utils/_contextlib.py", line 124, in decorate_context
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] return func(*args, **kwargs)
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/worker/gpu_model_runner.py", line 4307, in _dummy_sampler_run
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] raise RuntimeError(
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ...<4 lines>...
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] ) from e
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826] RuntimeError: CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.
+[0;36m(Worker_TP1 pid=75201)[0;0m ERROR 12-19 17:08:43 [multiproc_executor.py:826]
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] EngineCore failed to start.
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 857, in run_engine_core
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] engine_core = EngineCoreProc(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 637, in __init__
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] super().__init__(
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] vllm_config, executor_class, log_stats, executor_fail_callback
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 109, in __init__
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] num_gpu_blocks, num_cpu_blocks, kv_cache_config = self._initialize_kv_caches(
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] vllm_config
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 256, in _initialize_kv_caches
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] self.model_executor.initialize_from_config(kv_cache_configs)
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/abstract.py", line 116, in initialize_from_config
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] self.collective_rpc("compile_or_warm_up_model")
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 344, in get_response
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] raise RuntimeError(
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] )
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ERROR 12-19 17:08:43 [core.py:866] RuntimeError: Worker failed with error 'CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.', please check the stack trace above for the root cause
+[0;36m(EngineCore_DP0 pid=75118)[0;0m Process EngineCore_DP0:
+[0;36m(EngineCore_DP0 pid=75118)[0;0m Traceback (most recent call last):
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 313, in _bootstrap
+[0;36m(EngineCore_DP0 pid=75118)[0;0m self.run()
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/usr/lib64/python3.13/multiprocessing/process.py", line 108, in run
+[0;36m(EngineCore_DP0 pid=75118)[0;0m self._target(*self._args, **self._kwargs)
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 870, in run_engine_core
+[0;36m(EngineCore_DP0 pid=75118)[0;0m raise e
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 857, in run_engine_core
+[0;36m(EngineCore_DP0 pid=75118)[0;0m engine_core = EngineCoreProc(*args, **kwargs)
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 637, in __init__
+[0;36m(EngineCore_DP0 pid=75118)[0;0m super().__init__(
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m vllm_config, executor_class, log_stats, executor_fail_callback
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m )
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 109, in __init__
+[0;36m(EngineCore_DP0 pid=75118)[0;0m num_gpu_blocks, num_cpu_blocks, kv_cache_config = self._initialize_kv_caches(
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m vllm_config
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m )
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core.py", line 256, in _initialize_kv_caches
+[0;36m(EngineCore_DP0 pid=75118)[0;0m self.model_executor.initialize_from_config(kv_cache_configs)
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/abstract.py", line 116, in initialize_from_config
+[0;36m(EngineCore_DP0 pid=75118)[0;0m self.collective_rpc("compile_or_warm_up_model")
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 361, in collective_rpc
+[0;36m(EngineCore_DP0 pid=75118)[0;0m return aggregate(get_response())
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ~~~~~~~~~~~~^^
+[0;36m(EngineCore_DP0 pid=75118)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/executor/multiproc_executor.py", line 344, in get_response
+[0;36m(EngineCore_DP0 pid=75118)[0;0m raise RuntimeError(
+[0;36m(EngineCore_DP0 pid=75118)[0;0m ...<2 lines>...
+[0;36m(EngineCore_DP0 pid=75118)[0;0m )
+[0;36m(EngineCore_DP0 pid=75118)[0;0m RuntimeError: Worker failed with error 'CUDA out of memory occurred when warming up sampler with 64 dummy requests. Please try lowering `max_num_seqs` or `gpu_memory_utilization` when initializing the engine.', please check the stack trace above for the root cause
+[0;36m(Worker_TP0 pid=75200)[0;0m INFO 12-19 17:08:43 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(Worker_TP1 pid=75201)[0;0m INFO 12-19 17:08:43 [multiproc_executor.py:711] Parent process exited, terminating worker
+[0;36m(APIServer pid=74952)[0;0m Traceback (most recent call last):
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/bin/vllm", line 7, in
+[0;36m(APIServer pid=74952)[0;0m sys.exit(main())
+[0;36m(APIServer pid=74952)[0;0m ~~~~^^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/cli/main.py", line 73, in main
+[0;36m(APIServer pid=74952)[0;0m args.dispatch_function(args)
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/cli/serve.py", line 60, in cmd
+[0;36m(APIServer pid=74952)[0;0m uvloop.run(run_server(args))
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/uvloop/__init__.py", line 96, in run
+[0;36m(APIServer pid=74952)[0;0m return __asyncio.run(
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~^
+[0;36m(APIServer pid=74952)[0;0m wrapper(),
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m ...<2 lines>...
+[0;36m(APIServer pid=74952)[0;0m **run_kwargs
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m )
+[0;36m(APIServer pid=74952)[0;0m ^
+[0;36m(APIServer pid=74952)[0;0m File "/usr/lib64/python3.13/asyncio/runners.py", line 195, in run
+[0;36m(APIServer pid=74952)[0;0m return runner.run(main)
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/usr/lib64/python3.13/asyncio/runners.py", line 118, in run
+[0;36m(APIServer pid=74952)[0;0m return self._loop.run_until_complete(task)
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/uvloop/__init__.py", line 48, in wrapper
+[0;36m(APIServer pid=74952)[0;0m return await main
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 1398, in run_server
+[0;36m(APIServer pid=74952)[0;0m await run_server_worker(listen_address, sock, args, **uvicorn_kwargs)
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 1417, in run_server_worker
+[0;36m(APIServer pid=74952)[0;0m async with build_async_engine_client(
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=74952)[0;0m args,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^
+[0;36m(APIServer pid=74952)[0;0m client_config=client_config,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m ) as engine_client:
+[0;36m(APIServer pid=74952)[0;0m ^
+[0;36m(APIServer pid=74952)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 214, in __aenter__
+[0;36m(APIServer pid=74952)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 172, in build_async_engine_client
+[0;36m(APIServer pid=74952)[0;0m async with build_async_engine_client_from_engine_args(
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=74952)[0;0m engine_args,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m ...<2 lines>...
+[0;36m(APIServer pid=74952)[0;0m client_config=client_config,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m ) as engine:
+[0;36m(APIServer pid=74952)[0;0m ^
+[0;36m(APIServer pid=74952)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 214, in __aenter__
+[0;36m(APIServer pid=74952)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/entrypoints/openai/api_server.py", line 213, in build_async_engine_client_from_engine_args
+[0;36m(APIServer pid=74952)[0;0m async_llm = AsyncLLM.from_vllm_config(
+[0;36m(APIServer pid=74952)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=74952)[0;0m ...<6 lines>...
+[0;36m(APIServer pid=74952)[0;0m client_index=client_index,
+[0;36m(APIServer pid=74952)[0;0m )
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 215, in from_vllm_config
+[0;36m(APIServer pid=74952)[0;0m return cls(
+[0;36m(APIServer pid=74952)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=74952)[0;0m ...<9 lines>...
+[0;36m(APIServer pid=74952)[0;0m client_index=client_index,
+[0;36m(APIServer pid=74952)[0;0m )
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/async_llm.py", line 134, in __init__
+[0;36m(APIServer pid=74952)[0;0m self.engine_core = EngineCoreClient.make_async_mp_client(
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=74952)[0;0m vllm_config=vllm_config,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m ...<4 lines>...
+[0;36m(APIServer pid=74952)[0;0m client_index=client_index,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m )
+[0;36m(APIServer pid=74952)[0;0m ^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 121, in make_async_mp_client
+[0;36m(APIServer pid=74952)[0;0m return AsyncMPClient(*client_args)
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 820, in __init__
+[0;36m(APIServer pid=74952)[0;0m super().__init__(
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=74952)[0;0m asyncio_mode=True,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m ...<3 lines>...
+[0;36m(APIServer pid=74952)[0;0m client_addresses=client_addresses,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m )
+[0;36m(APIServer pid=74952)[0;0m ^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/core_client.py", line 477, in __init__
+[0;36m(APIServer pid=74952)[0;0m with launch_core_engines(vllm_config, executor_class, log_stats) as (
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/usr/lib64/python3.13/contextlib.py", line 148, in __exit__
+[0;36m(APIServer pid=74952)[0;0m next(self.gen)
+[0;36m(APIServer pid=74952)[0;0m ~~~~^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/utils.py", line 903, in launch_core_engines
+[0;36m(APIServer pid=74952)[0;0m wait_for_engine_startup(
+[0;36m(APIServer pid=74952)[0;0m ~~~~~~~~~~~~~~~~~~~~~~~^
+[0;36m(APIServer pid=74952)[0;0m handshake_socket,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m ...<5 lines>...
+[0;36m(APIServer pid=74952)[0;0m coordinator.proc if coordinator else None,
+[0;36m(APIServer pid=74952)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=74952)[0;0m )
+[0;36m(APIServer pid=74952)[0;0m ^
+[0;36m(APIServer pid=74952)[0;0m File "/opt/venv/lib64/python3.13/site-packages/vllm/v1/engine/utils.py", line 960, in wait_for_engine_startup
+[0;36m(APIServer pid=74952)[0;0m raise RuntimeError(
+[0;36m(APIServer pid=74952)[0;0m ...<3 lines>...
+[0;36m(APIServer pid=74952)[0;0m )
+[0;36m(APIServer pid=74952)[0;0m RuntimeError: Engine core initialization failed. See root cause above. Failed core proc(s): {}
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..acdb5fe
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-e4e803cf-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 200.63 \nTotal input tokens: 38358 \nTotal generated tokens: 40296 \nRequest throughput (req/s): 0.90 \nOutput token throughput (tok/s): 200.85 \nPeak output token throughput (tok/s): 350.00 \nPeak concurrent requests: 16.00 \nTotal Token throughput (tok/s): 392.04 \n---------------Time to First Token----------------\nMean TTFT (ms): 100.59 \nMedian TTFT (ms): 84.87 \nP99 TTFT (ms): 217.06 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 41.46 \nMedian TPOT (ms): 41.30 \nP99 TPOT (ms): 45.56 \n---------------Inter-token Latency----------------\nMean ITL (ms): 41.46 \nMedian ITL (ms): 40.41 \nP99 ITL (ms): 81.46 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..fe1d0b5
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-d2d3278b-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 211.85 \nTotal input tokens: 146694 \nTotal generated tokens: 155558 \nRequest throughput (req/s): 3.40 \nOutput token throughput (tok/s): 734.30 \nPeak output token throughput (tok/s): 1241.00 \nPeak concurrent requests: 68.00 \nTotal Token throughput (tok/s): 1426.75 \n---------------Time to First Token----------------\nMean TTFT (ms): 111.24 \nMedian TTFT (ms): 89.87 \nP99 TTFT (ms): 307.29 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 52.80 \nMedian TPOT (ms): 52.89 \nP99 TPOT (ms): 76.85 \n---------------Inter-token Latency----------------\nMean ITL (ms): 52.36 \nMedian ITL (ms): 48.02 \nP99 ITL (ms): 172.90 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..a5a823a
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1042 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:38:31 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:38:31 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=15688)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:38:35 [model.py:629] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:38:35 [model.py:1755] Using max model len 32768
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:38:35 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:39 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:39 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:56369 backend=nccl
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:39 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:39 [gpu_model_runner.py:3544] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:40 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 4.28it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.55it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.26it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.21it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.33it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:43 [default_loader.py:308] Loading weights took 3.08 seconds
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:44 [gpu_model_runner.py:3626] Model loading took 15.4180 GiB memory and 3.812286 seconds
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:49 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/1372a8f1b2/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:49 [backends.py:676] Dynamo bytecode transform time: 5.17 s
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:38:55 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:39:19 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 27.34 s
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:39:19 [monitor.py:34] torch.compile takes 32.51 s in total
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:39:21 [gpu_worker.py:364] Available KV cache memory: 13.76 GiB
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:39:22 [kv_cache_utils.py:1287] GPU KV cache size: 90,160 tokens
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:39:22 [kv_cache_utils.py:1292] Maximum concurrency for 32,768 tokens per request: 2.75x
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 17.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 20.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 20.85it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 20.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 20.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 21.07it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 20.83it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:00, 18.50it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:00<00:00, 20.67it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 20.73it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 21.18it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 20.88it/s]
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:39:24 [gpu_model_runner.py:4548] Graph capturing finished in 2 secs, took 1.61 GiB
+[0;36m(EngineCore_DP0 pid=15850)[0;0m INFO 12-12 17:39:24 [core.py:256] init engine (profile, create kv cache, warmup model) took 40.43 seconds
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=15688)[0;0m WARNING 12-12 17:39:26 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [serving_responses.py:197] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:26 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=15688)[0;0m INFO: Started server process [15688]
+[0;36m(APIServer pid=15688)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=15688)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53572 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:46 [loggers.py:248] Engine 000: Avg prompt throughput: 8.3 tokens/s, Avg generation throughput: 28.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:39:56 [loggers.py:248] Engine 000: Avg prompt throughput: 170.5 tokens/s, Avg generation throughput: 115.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:40:06 [loggers.py:248] Engine 000: Avg prompt throughput: 249.4 tokens/s, Avg generation throughput: 172.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:40:16 [loggers.py:248] Engine 000: Avg prompt throughput: 214.4 tokens/s, Avg generation throughput: 207.3 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:40:26 [loggers.py:248] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 176.2 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:40:36 [loggers.py:248] Engine 000: Avg prompt throughput: 379.7 tokens/s, Avg generation throughput: 213.2 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:40:46 [loggers.py:248] Engine 000: Avg prompt throughput: 400.9 tokens/s, Avg generation throughput: 230.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:40:56 [loggers.py:248] Engine 000: Avg prompt throughput: 215.3 tokens/s, Avg generation throughput: 199.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:41:06 [loggers.py:248] Engine 000: Avg prompt throughput: 298.4 tokens/s, Avg generation throughput: 221.9 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:41:16 [loggers.py:248] Engine 000: Avg prompt throughput: 301.7 tokens/s, Avg generation throughput: 290.1 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:41:26 [loggers.py:248] Engine 000: Avg prompt throughput: 160.7 tokens/s, Avg generation throughput: 310.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:41:36 [loggers.py:248] Engine 000: Avg prompt throughput: 36.4 tokens/s, Avg generation throughput: 243.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:41:46 [loggers.py:248] Engine 000: Avg prompt throughput: 232.2 tokens/s, Avg generation throughput: 194.9 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:41:56 [loggers.py:248] Engine 000: Avg prompt throughput: 219.2 tokens/s, Avg generation throughput: 242.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:38512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:42:06 [loggers.py:248] Engine 000: Avg prompt throughput: 157.6 tokens/s, Avg generation throughput: 242.6 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:42:16 [loggers.py:248] Engine 000: Avg prompt throughput: 65.8 tokens/s, Avg generation throughput: 149.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:51062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:42:26 [loggers.py:248] Engine 000: Avg prompt throughput: 180.0 tokens/s, Avg generation throughput: 183.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:35248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:42860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:42:36 [loggers.py:248] Engine 000: Avg prompt throughput: 189.6 tokens/s, Avg generation throughput: 176.4 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:41458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:42:46 [loggers.py:248] Engine 000: Avg prompt throughput: 139.6 tokens/s, Avg generation throughput: 256.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:42:56 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 143.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:43:06 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 42.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:43:16 [loggers.py:248] Engine 000: Avg prompt throughput: 137.1 tokens/s, Avg generation throughput: 49.3 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 3.2%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:43:26 [loggers.py:248] Engine 000: Avg prompt throughput: 1062.8 tokens/s, Avg generation throughput: 406.1 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.5%, Prefix cache hit rate: 22.9%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:43:36 [loggers.py:248] Engine 000: Avg prompt throughput: 1072.7 tokens/s, Avg generation throughput: 742.0 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.8%, Prefix cache hit rate: 35.7%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:43:46 [loggers.py:248] Engine 000: Avg prompt throughput: 825.1 tokens/s, Avg generation throughput: 825.9 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.6%, Prefix cache hit rate: 42.9%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:43:56 [loggers.py:248] Engine 000: Avg prompt throughput: 439.4 tokens/s, Avg generation throughput: 894.3 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.2%, Prefix cache hit rate: 46.1%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:44:06 [loggers.py:248] Engine 000: Avg prompt throughput: 1004.6 tokens/s, Avg generation throughput: 782.2 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.1%, Prefix cache hit rate: 43.9%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:44:16 [loggers.py:248] Engine 000: Avg prompt throughput: 1023.4 tokens/s, Avg generation throughput: 865.6 tokens/s, Running: 55 reqs, Waiting: 0 reqs, GPU KV cache usage: 22.8%, Prefix cache hit rate: 39.2%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:44:26 [loggers.py:248] Engine 000: Avg prompt throughput: 551.7 tokens/s, Avg generation throughput: 921.0 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.9%, Prefix cache hit rate: 37.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:44:36 [loggers.py:248] Engine 000: Avg prompt throughput: 891.5 tokens/s, Avg generation throughput: 873.5 tokens/s, Running: 58 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.0%, Prefix cache hit rate: 33.9%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:44:46 [loggers.py:248] Engine 000: Avg prompt throughput: 652.9 tokens/s, Avg generation throughput: 1085.5 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.4%, Prefix cache hit rate: 32.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:44:56 [loggers.py:248] Engine 000: Avg prompt throughput: 907.6 tokens/s, Avg generation throughput: 903.2 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.9%, Prefix cache hit rate: 29.7%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:55674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:45:06 [loggers.py:248] Engine 000: Avg prompt throughput: 916.0 tokens/s, Avg generation throughput: 751.3 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.5%, Prefix cache hit rate: 27.6%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:45:16 [loggers.py:248] Engine 000: Avg prompt throughput: 471.8 tokens/s, Avg generation throughput: 760.2 tokens/s, Running: 36 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.6%, Prefix cache hit rate: 26.7%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:33978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:45:26 [loggers.py:248] Engine 000: Avg prompt throughput: 847.9 tokens/s, Avg generation throughput: 755.8 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.8%, Prefix cache hit rate: 25.3%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:33988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:33978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:33988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:45:36 [loggers.py:248] Engine 000: Avg prompt throughput: 971.8 tokens/s, Avg generation throughput: 775.8 tokens/s, Running: 34 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.6%, Prefix cache hit rate: 23.7%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:45:46 [loggers.py:248] Engine 000: Avg prompt throughput: 509.6 tokens/s, Avg generation throughput: 749.1 tokens/s, Running: 40 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.8%, Prefix cache hit rate: 23.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:33988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:45:56 [loggers.py:248] Engine 000: Avg prompt throughput: 741.8 tokens/s, Avg generation throughput: 780.6 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.4%, Prefix cache hit rate: 22.0%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:53984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:33988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:34020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:46:06 [loggers.py:248] Engine 000: Avg prompt throughput: 949.2 tokens/s, Avg generation throughput: 797.3 tokens/s, Running: 45 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.1%, Prefix cache hit rate: 20.8%
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:48266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:40332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:50216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:36258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:54038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:45046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO: 127.0.0.1:37514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:46:16 [loggers.py:248] Engine 000: Avg prompt throughput: 692.8 tokens/s, Avg generation throughput: 925.9 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.7%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:46:26 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 632.4 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.6%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:46:36 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 210.0 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.5%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=15688)[0;0m INFO 12-12 17:46:45 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1212 17:46:45.021549801 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..8ac2807
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 499.5056346299998,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 2.00197941859201,
+ "tokens_per_second": 1484.1354103024892
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
new file mode 100644
index 0000000..9aa183b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 680.202111824,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 1.4701512721247574,
+ "tokens_per_second": 1089.8731231693348
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..1850560
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-fce2ee75-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 195.29 \nTotal input tokens: 37841 \nTotal generated tokens: 38821 \nRequest throughput (req/s): 0.92 \nOutput token throughput (tok/s): 198.79 \nPeak output token throughput (tok/s): 406.00 \nPeak concurrent requests: 13.00 \nTotal Token throughput (tok/s): 392.56 \n---------------Time to First Token----------------\nMean TTFT (ms): 73.39 \nMedian TTFT (ms): 61.33 \nP99 TTFT (ms): 185.43 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 30.73 \nMedian TPOT (ms): 30.60 \nP99 TPOT (ms): 33.54 \n---------------Inter-token Latency----------------\nMean ITL (ms): 30.74 \nMedian ITL (ms): 29.96 \nP99 ITL (ms): 50.56 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..d9912a2
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-af5860ff-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 204.38 \nTotal input tokens: 145810 \nTotal generated tokens: 152213 \nRequest throughput (req/s): 3.52 \nOutput token throughput (tok/s): 744.74 \nPeak output token throughput (tok/s): 1238.00 \nPeak concurrent requests: 53.00 \nTotal Token throughput (tok/s): 1458.16 \n---------------Time to First Token----------------\nMean TTFT (ms): 75.15 \nMedian TTFT (ms): 63.63 \nP99 TTFT (ms): 179.56 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 35.83 \nMedian TPOT (ms): 35.55 \nP99 TPOT (ms): 48.38 \n---------------Inter-token Latency----------------\nMean ITL (ms): 35.60 \nMedian ITL (ms): 33.55 \nP99 ITL (ms): 112.07 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
new file mode 100644
index 0000000..b7dfd8d
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
@@ -0,0 +1,1040 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:58:44 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:58:44 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 65536, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:58:48 [model.py:629] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:58:48 [model.py:1755] Using max model len 65536
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:58:48 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:52 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=65536, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:52 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:38471 backend=nccl
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:52 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:52 [gpu_model_runner.py:3544] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:52 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 5.67it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.57it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.23it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.15it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.28it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:57 [default_loader.py:308] Loading weights took 3.12 seconds
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:57 [gpu_model_runner.py:3626] Model loading took 15.0586 GiB memory and 4.185534 seconds
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:59 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/dfa108d3b7/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:58:59 [backends.py:676] Dynamo bytecode transform time: 2.05 s
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:01 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:02 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 1.22 s
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:02 [monitor.py:34] torch.compile takes 3.27 s in total
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:03 [gpu_worker.py:364] Available KV cache memory: 15.33 GiB
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:03 [kv_cache_utils.py:1287] GPU KV cache size: 125,616 tokens
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:03 [kv_cache_utils.py:1292] Maximum concurrency for 65,536 tokens per request: 1.92x
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:00, 22.23it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 23.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:00<00:00, 23.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 24.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:00<00:00, 25.10it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:00<00:00, 25.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 24.93it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 22.74it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 25.42it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 25.45it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 25.35it/s]
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:05 [gpu_model_runner.py:4548] Graph capturing finished in 2 secs, took 0.94 GiB
+[0;36m(EngineCore_DP0 pid=8607)[0;0m INFO 12-12 16:59:05 [core.py:256] init engine (profile, create kv cache, warmup model) took 8.08 seconds
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=8444)[0;0m WARNING 12-12 16:59:07 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [serving_responses.py:197] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:07 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=8444)[0;0m INFO: Started server process [8444]
+[0;36m(APIServer pid=8444)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=8444)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:45004 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:27 [loggers.py:248] Engine 000: Avg prompt throughput: 41.7 tokens/s, Avg generation throughput: 40.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:37 [loggers.py:248] Engine 000: Avg prompt throughput: 135.3 tokens/s, Avg generation throughput: 125.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:47 [loggers.py:248] Engine 000: Avg prompt throughput: 183.3 tokens/s, Avg generation throughput: 213.5 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 16:59:57 [loggers.py:248] Engine 000: Avg prompt throughput: 278.0 tokens/s, Avg generation throughput: 151.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:00:07 [loggers.py:248] Engine 000: Avg prompt throughput: 191.1 tokens/s, Avg generation throughput: 189.3 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:00:17 [loggers.py:248] Engine 000: Avg prompt throughput: 365.3 tokens/s, Avg generation throughput: 210.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:00:27 [loggers.py:248] Engine 000: Avg prompt throughput: 365.3 tokens/s, Avg generation throughput: 226.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:50366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:00:37 [loggers.py:248] Engine 000: Avg prompt throughput: 149.5 tokens/s, Avg generation throughput: 197.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:00:47 [loggers.py:248] Engine 000: Avg prompt throughput: 376.3 tokens/s, Avg generation throughput: 238.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:00:57 [loggers.py:248] Engine 000: Avg prompt throughput: 223.3 tokens/s, Avg generation throughput: 301.6 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:01:07 [loggers.py:248] Engine 000: Avg prompt throughput: 276.0 tokens/s, Avg generation throughput: 299.4 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:01:17 [loggers.py:248] Engine 000: Avg prompt throughput: 36.1 tokens/s, Avg generation throughput: 191.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:01:27 [loggers.py:248] Engine 000: Avg prompt throughput: 238.5 tokens/s, Avg generation throughput: 194.0 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:01:37 [loggers.py:248] Engine 000: Avg prompt throughput: 274.2 tokens/s, Avg generation throughput: 248.1 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:01:47 [loggers.py:248] Engine 000: Avg prompt throughput: 97.1 tokens/s, Avg generation throughput: 244.7 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:01:57 [loggers.py:248] Engine 000: Avg prompt throughput: 94.5 tokens/s, Avg generation throughput: 152.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:02:07 [loggers.py:248] Engine 000: Avg prompt throughput: 106.3 tokens/s, Avg generation throughput: 144.6 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:38744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:02:17 [loggers.py:248] Engine 000: Avg prompt throughput: 230.9 tokens/s, Avg generation throughput: 195.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:46354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:39080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:59740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:02:27 [loggers.py:248] Engine 000: Avg prompt throughput: 122.7 tokens/s, Avg generation throughput: 248.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:02:37 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 78.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:02:47 [loggers.py:248] Engine 000: Avg prompt throughput: 1.3 tokens/s, Avg generation throughput: 13.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:02:57 [loggers.py:248] Engine 000: Avg prompt throughput: 644.5 tokens/s, Avg generation throughput: 299.5 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 13.9%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:03:07 [loggers.py:248] Engine 000: Avg prompt throughput: 1250.8 tokens/s, Avg generation throughput: 730.9 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.4%, Prefix cache hit rate: 32.1%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:03:17 [loggers.py:248] Engine 000: Avg prompt throughput: 733.4 tokens/s, Avg generation throughput: 852.0 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.3%, Prefix cache hit rate: 39.3%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:03:27 [loggers.py:248] Engine 000: Avg prompt throughput: 731.3 tokens/s, Avg generation throughput: 931.6 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.3%, Prefix cache hit rate: 45.2%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:03:37 [loggers.py:248] Engine 000: Avg prompt throughput: 905.5 tokens/s, Avg generation throughput: 759.1 tokens/s, Running: 34 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.9%, Prefix cache hit rate: 45.1%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:03:47 [loggers.py:248] Engine 000: Avg prompt throughput: 885.9 tokens/s, Avg generation throughput: 953.7 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.1%, Prefix cache hit rate: 40.7%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:03:57 [loggers.py:248] Engine 000: Avg prompt throughput: 752.8 tokens/s, Avg generation throughput: 868.0 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 37.5%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:04:07 [loggers.py:248] Engine 000: Avg prompt throughput: 637.8 tokens/s, Avg generation throughput: 781.9 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.6%, Prefix cache hit rate: 35.2%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:04:17 [loggers.py:248] Engine 000: Avg prompt throughput: 770.5 tokens/s, Avg generation throughput: 1128.0 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.8%, Prefix cache hit rate: 32.7%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:04:27 [loggers.py:248] Engine 000: Avg prompt throughput: 619.8 tokens/s, Avg generation throughput: 908.0 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.8%, Prefix cache hit rate: 31.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:04:37 [loggers.py:248] Engine 000: Avg prompt throughput: 1039.1 tokens/s, Avg generation throughput: 753.8 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.5%, Prefix cache hit rate: 28.5%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:04:47 [loggers.py:248] Engine 000: Avg prompt throughput: 653.5 tokens/s, Avg generation throughput: 642.5 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.1%, Prefix cache hit rate: 27.1%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:04:57 [loggers.py:248] Engine 000: Avg prompt throughput: 758.8 tokens/s, Avg generation throughput: 678.4 tokens/s, Running: 29 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 25.6%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:05:07 [loggers.py:248] Engine 000: Avg prompt throughput: 746.7 tokens/s, Avg generation throughput: 898.2 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.0%, Prefix cache hit rate: 24.5%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:05:17 [loggers.py:248] Engine 000: Avg prompt throughput: 828.3 tokens/s, Avg generation throughput: 658.6 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.7%, Prefix cache hit rate: 23.2%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:05:27 [loggers.py:248] Engine 000: Avg prompt throughput: 582.5 tokens/s, Avg generation throughput: 758.8 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.8%, Prefix cache hit rate: 22.4%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:54998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:05:37 [loggers.py:248] Engine 000: Avg prompt throughput: 644.3 tokens/s, Avg generation throughput: 765.5 tokens/s, Running: 32 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.8%, Prefix cache hit rate: 21.5%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:47568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:43478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:33706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:35606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:43494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:05:47 [loggers.py:248] Engine 000: Avg prompt throughput: 1325.2 tokens/s, Avg generation throughput: 877.6 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.9%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:55206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:05:57 [loggers.py:248] Engine 000: Avg prompt throughput: 69.6 tokens/s, Avg generation throughput: 709.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.3%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:06:07 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 226.4 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=8444)[0;0m INFO 12-12 17:06:12 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1212 17:06:12.195241843 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
new file mode 100644
index 0000000..46cdb87
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 354.4166975950002,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 2.821537491844479,
+ "tokens_per_second": 2077.5827013698454
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..9192f92
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-1dd28fd7-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 214.85 \nTotal input tokens: 38756 \nTotal generated tokens: 39194 \nRequest throughput (req/s): 0.84 \nOutput token throughput (tok/s): 182.43 \nPeak output token throughput (tok/s): 308.00 \nPeak concurrent requests: 19.00 \nTotal Token throughput (tok/s): 362.82 \n---------------Time to First Token----------------\nMean TTFT (ms): 115.51 \nMedian TTFT (ms): 108.55 \nP99 TTFT (ms): 235.09 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 53.14 \nMedian TPOT (ms): 53.86 \nP99 TPOT (ms): 69.03 \n---------------Inter-token Latency----------------\nMean ITL (ms): 53.53 \nMedian ITL (ms): 52.81 \nP99 ITL (ms): 120.56 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..788910c
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-2f4f4694-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 236.39 \nTotal input tokens: 145540 \nTotal generated tokens: 152178 \nRequest throughput (req/s): 3.05 \nOutput token throughput (tok/s): 643.77 \nPeak output token throughput (tok/s): 1024.00 \nPeak concurrent requests: 92.00 \nTotal Token throughput (tok/s): 1259.45 \n---------------Time to First Token----------------\nMean TTFT (ms): 850.20 \nMedian TTFT (ms): 141.29 \nP99 TTFT (ms): 5024.92 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 71.13 \nMedian TPOT (ms): 71.70 \nP99 TPOT (ms): 89.98 \n---------------Inter-token Latency----------------\nMean ITL (ms): 70.90 \nMedian ITL (ms): 66.80 \nP99 ITL (ms): 157.64 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_server.log b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_server.log
new file mode 100644
index 0000000..6ea9f83
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_server.log
@@ -0,0 +1,1045 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:18:27 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:18:27 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=13351)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:18:31 [model.py:629] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=13351)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 7.46it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 20.07it/s]
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:18:32 [model.py:1755] Using max model len 32768
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:18:32 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:18:32 [config.py:269] Overriding max cuda graph capture size to 1024 for performance.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:36 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:36 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:47683 backend=nccl
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:36 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:36 [gpu_model_runner.py:3544] Starting to load model openai/gpt-oss-20b...
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:36 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:36 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:36 [mxfp4.py:171] Using Triton backend
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:01, 1.43it/s]
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:01<00:00, 1.11it/s]
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.05it/s]
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.09it/s]
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:40 [default_loader.py:308] Loading weights took 2.82 seconds
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:40 [gpu_model_runner.py:3626] Model loading took 14.3066 GiB memory and 3.521807 seconds
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:42 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/fd3d592b37/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:42 [backends.py:676] Dynamo bytecode transform time: 1.70 s
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:18:44 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:19:02 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 18.91 s
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:19:02 [monitor.py:34] torch.compile takes 20.61 s in total
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:19:04 [gpu_worker.py:364] Available KV cache memory: 15.30 GiB
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:19:04 [kv_cache_utils.py:1287] GPU KV cache size: 334,208 tokens
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:19:04 [kv_cache_utils.py:1292] Maximum concurrency for 32,768 tokens per request: 19.12x
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 1%| | 1/83 [00:00<00:14, 5.74it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:13, 6.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 4%|▎ | 3/83 [00:00<00:12, 6.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:00<00:12, 6.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 6%|▌ | 5/83 [00:00<00:12, 6.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:00<00:11, 6.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 8%|▊ | 7/83 [00:01<00:11, 6.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:01<00:11, 6.70it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 9/83 [00:01<00:10, 6.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:01<00:10, 6.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 13%|█▎ | 11/83 [00:01<00:10, 6.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:01<00:10, 7.05it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 13/83 [00:01<00:09, 7.13it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:02<00:09, 7.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 15/83 [00:02<00:09, 7.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:02<00:09, 7.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 20%|██ | 17/83 [00:02<00:08, 7.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:02<00:08, 7.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 23%|██▎ | 19/83 [00:02<00:08, 7.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:02<00:08, 7.74it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 25%|██▌ | 21/83 [00:02<00:07, 7.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:03<00:07, 8.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 28%|██▊ | 23/83 [00:03<00:07, 8.04it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:03<00:07, 8.16it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 30%|███ | 25/83 [00:03<00:06, 8.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:03<00:06, 8.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 33%|███▎ | 27/83 [00:03<00:06, 8.61it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:03<00:06, 8.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 35%|███▍ | 29/83 [00:03<00:05, 9.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 31/83 [00:04<00:05, 9.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 40%|███▉ | 33/83 [00:04<00:05, 9.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 35/83 [00:04<00:04, 10.04it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▍ | 37/83 [00:04<00:04, 10.38it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 39/83 [00:04<00:04, 10.69it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 49%|████▉ | 41/83 [00:04<00:03, 11.04it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 52%|█████▏ | 43/83 [00:05<00:03, 11.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 54%|█████▍ | 45/83 [00:05<00:03, 11.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 57%|█████▋ | 47/83 [00:05<00:02, 12.62it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 59%|█████▉ | 49/83 [00:05<00:02, 13.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 61%|██████▏ | 51/83 [00:05<00:02, 13.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▍ | 53/83 [00:05<00:02, 14.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 66%|██████▋ | 55/83 [00:05<00:01, 14.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 69%|██████▊ | 57/83 [00:06<00:01, 15.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 71%|███████ | 59/83 [00:06<00:01, 16.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 61/83 [00:06<00:01, 16.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 76%|███████▌ | 63/83 [00:06<00:01, 17.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 78%|███████▊ | 65/83 [00:06<00:01, 17.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:06<00:00, 19.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 86%|████████▌ | 71/83 [00:06<00:00, 20.07it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:06<00:00, 21.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 93%|█████████▎| 77/83 [00:07<00:00, 22.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 96%|█████████▋| 80/83 [00:07<00:00, 23.94it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:07<00:00, 19.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:07<00:00, 11.30it/s]
+[0;36m(EngineCore_DP0 pid=13518)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 24.85it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 26.95it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 23.53it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 21.90it/s]
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:19:12 [gpu_model_runner.py:4548] Graph capturing finished in 9 secs, took 0.82 GiB
+[0;36m(EngineCore_DP0 pid=13518)[0;0m INFO 12-12 17:19:13 [core.py:256] init engine (profile, create kv cache, warmup model) took 32.26 seconds
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:14 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=13351)[0;0m WARNING 12-12 17:19:14 [serving_responses.py:218] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:21 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=13351)[0;0m INFO: Started server process [13351]
+[0;36m(APIServer pid=13351)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=13351)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33010 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:41 [loggers.py:248] Engine 000: Avg prompt throughput: 7.5 tokens/s, Avg generation throughput: 19.0 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:19:51 [loggers.py:248] Engine 000: Avg prompt throughput: 166.7 tokens/s, Avg generation throughput: 106.5 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:20:01 [loggers.py:248] Engine 000: Avg prompt throughput: 250.6 tokens/s, Avg generation throughput: 150.2 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:20:11 [loggers.py:248] Engine 000: Avg prompt throughput: 211.3 tokens/s, Avg generation throughput: 187.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:20:21 [loggers.py:248] Engine 000: Avg prompt throughput: 225.3 tokens/s, Avg generation throughput: 185.4 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:20:31 [loggers.py:248] Engine 000: Avg prompt throughput: 370.0 tokens/s, Avg generation throughput: 198.3 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:20:41 [loggers.py:248] Engine 000: Avg prompt throughput: 389.0 tokens/s, Avg generation throughput: 222.6 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:20:51 [loggers.py:248] Engine 000: Avg prompt throughput: 212.4 tokens/s, Avg generation throughput: 203.7 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:21:01 [loggers.py:248] Engine 000: Avg prompt throughput: 291.9 tokens/s, Avg generation throughput: 207.8 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:21:11 [loggers.py:248] Engine 000: Avg prompt throughput: 325.2 tokens/s, Avg generation throughput: 251.0 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:56640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:56640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:54080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:21:21 [loggers.py:248] Engine 000: Avg prompt throughput: 223.0 tokens/s, Avg generation throughput: 284.9 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:56640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:32898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:21:31 [loggers.py:248] Engine 000: Avg prompt throughput: 44.0 tokens/s, Avg generation throughput: 278.4 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:32898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:48108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:21:41 [loggers.py:248] Engine 000: Avg prompt throughput: 225.8 tokens/s, Avg generation throughput: 183.5 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:48110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:48120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:48110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:48108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:21:51 [loggers.py:248] Engine 000: Avg prompt throughput: 214.7 tokens/s, Avg generation throughput: 257.4 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:48120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:32898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:56640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:22:01 [loggers.py:248] Engine 000: Avg prompt throughput: 153.7 tokens/s, Avg generation throughput: 218.6 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:32898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:22:11 [loggers.py:248] Engine 000: Avg prompt throughput: 63.4 tokens/s, Avg generation throughput: 170.1 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:32898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:22:21 [loggers.py:248] Engine 000: Avg prompt throughput: 175.7 tokens/s, Avg generation throughput: 178.3 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:36912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:22:31 [loggers.py:248] Engine 000: Avg prompt throughput: 72.8 tokens/s, Avg generation throughput: 179.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:58540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:22:41 [loggers.py:248] Engine 000: Avg prompt throughput: 253.8 tokens/s, Avg generation throughput: 226.5 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:22:51 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 165.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:23:01 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 28.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:23:11 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 26.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:23:21 [loggers.py:248] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 9.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:23:31 [loggers.py:248] Engine 000: Avg prompt throughput: 598.9 tokens/s, Avg generation throughput: 165.8 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 12.9%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:23:41 [loggers.py:248] Engine 000: Avg prompt throughput: 1232.7 tokens/s, Avg generation throughput: 473.2 tokens/s, Running: 38 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 30.9%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:23:51 [loggers.py:248] Engine 000: Avg prompt throughput: 875.5 tokens/s, Avg generation throughput: 694.1 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 39.5%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:24:01 [loggers.py:248] Engine 000: Avg prompt throughput: 607.8 tokens/s, Avg generation throughput: 759.7 tokens/s, Running: 54 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 44.3%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:24:11 [loggers.py:248] Engine 000: Avg prompt throughput: 780.6 tokens/s, Avg generation throughput: 809.2 tokens/s, Running: 60 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 46.7%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:24:21 [loggers.py:248] Engine 000: Avg prompt throughput: 888.7 tokens/s, Avg generation throughput: 785.0 tokens/s, Running: 58 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 42.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:24:31 [loggers.py:248] Engine 000: Avg prompt throughput: 773.2 tokens/s, Avg generation throughput: 867.9 tokens/s, Running: 64 reqs, Waiting: 5 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 38.6%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:24:41 [loggers.py:248] Engine 000: Avg prompt throughput: 626.4 tokens/s, Avg generation throughput: 850.0 tokens/s, Running: 64 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 36.3%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:24:51 [loggers.py:248] Engine 000: Avg prompt throughput: 626.1 tokens/s, Avg generation throughput: 908.8 tokens/s, Running: 63 reqs, Waiting: 22 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 34.2%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:25:01 [loggers.py:248] Engine 000: Avg prompt throughput: 788.6 tokens/s, Avg generation throughput: 883.2 tokens/s, Running: 63 reqs, Waiting: 8 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 31.9%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:25:11 [loggers.py:248] Engine 000: Avg prompt throughput: 938.4 tokens/s, Avg generation throughput: 870.3 tokens/s, Running: 63 reqs, Waiting: 9 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 30.1%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:25:21 [loggers.py:248] Engine 000: Avg prompt throughput: 880.6 tokens/s, Avg generation throughput: 785.5 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 28.1%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:25:31 [loggers.py:248] Engine 000: Avg prompt throughput: 754.3 tokens/s, Avg generation throughput: 760.2 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 26.6%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:43958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:43964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:43958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:25:41 [loggers.py:248] Engine 000: Avg prompt throughput: 726.1 tokens/s, Avg generation throughput: 841.5 tokens/s, Running: 59 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 25.5%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:25:51 [loggers.py:248] Engine 000: Avg prompt throughput: 881.5 tokens/s, Avg generation throughput: 723.8 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 24.1%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:43964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:40992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:49796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:26:01 [loggers.py:248] Engine 000: Avg prompt throughput: 687.5 tokens/s, Avg generation throughput: 748.2 tokens/s, Running: 52 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 23.1%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:39206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:43958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:50160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:26:11 [loggers.py:248] Engine 000: Avg prompt throughput: 640.8 tokens/s, Avg generation throughput: 774.6 tokens/s, Running: 62 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 22.2%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:43964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:35900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:26:21 [loggers.py:248] Engine 000: Avg prompt throughput: 1011.4 tokens/s, Avg generation throughput: 803.5 tokens/s, Running: 64 reqs, Waiting: 6 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 21.0%
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:55318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:60958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:37400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:26:31 [loggers.py:248] Engine 000: Avg prompt throughput: 234.5 tokens/s, Avg generation throughput: 870.9 tokens/s, Running: 45 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:26:41 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 491.1 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:26:51 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 217.3 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:27:01 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 90.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:27:11 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 27.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=13351)[0;0m INFO 12-12 17:27:20 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1212 17:27:20.890022828 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_throughput.json
new file mode 100644
index 0000000..4e28b3b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700-uv+pl/openai_gpt-oss-20b_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 602.7616506659997,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 1.6590305619063292,
+ "tokens_per_second": 1225.6785068919007
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..f9e10f7
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-84569f25-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 203.97 \nTotal input tokens: 38358 \nTotal generated tokens: 40296 \nRequest throughput (req/s): 0.88 \nOutput token throughput (tok/s): 197.56 \nPeak output token throughput (tok/s): 330.00 \nPeak concurrent requests: 18.00 \nTotal Token throughput (tok/s): 385.62 \n---------------Time to First Token----------------\nMean TTFT (ms): 108.91 \nMedian TTFT (ms): 95.15 \nP99 TTFT (ms): 230.11 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 47.90 \nMedian TPOT (ms): 47.66 \nP99 TPOT (ms): 53.10 \n---------------Inter-token Latency----------------\nMean ITL (ms): 47.94 \nMedian ITL (ms): 46.64 \nP99 ITL (ms): 85.76 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..12f0a86
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-a714d425-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 217.26 \nTotal input tokens: 146694 \nTotal generated tokens: 155585 \nRequest throughput (req/s): 3.31 \nOutput token throughput (tok/s): 716.13 \nPeak output token throughput (tok/s): 1152.00 \nPeak concurrent requests: 77.00 \nTotal Token throughput (tok/s): 1391.33 \n---------------Time to First Token----------------\nMean TTFT (ms): 186.47 \nMedian TTFT (ms): 105.59 \nP99 TTFT (ms): 1603.40 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 59.68 \nMedian TPOT (ms): 59.91 \nP99 TPOT (ms): 84.43 \n---------------Inter-token Latency----------------\nMean ITL (ms): 59.35 \nMedian ITL (ms): 54.32 \nP99 ITL (ms): 190.63 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..7dc483e
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1047 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:31:49 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:31:49 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=11550)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:31:53 [model.py:629] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:31:53 [model.py:1755] Using max model len 32768
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:31:53 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:31:57 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:31:57 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:56787 backend=nccl
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:31:57 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:31:58 [gpu_model_runner.py:3544] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:31:58 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 4.01it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.56it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.27it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.18it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.30it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:02 [default_loader.py:308] Loading weights took 3.14 seconds
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:02 [gpu_model_runner.py:3626] Model loading took 15.4180 GiB memory and 3.928491 seconds
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:08 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/1372a8f1b2/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:08 [backends.py:676] Dynamo bytecode transform time: 5.15 s
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:14 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:38 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 27.87 s
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:38 [monitor.py:34] torch.compile takes 33.02 s in total
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:41 [gpu_worker.py:364] Available KV cache memory: 11.80 GiB
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:41 [kv_cache_utils.py:1287] GPU KV cache size: 77,312 tokens
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:41 [kv_cache_utils.py:1292] Maximum concurrency for 32,768 tokens per request: 2.36x
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:01, 16.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:00<00:00, 17.79it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 18.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 18.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:00<00:00, 18.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 18.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 18.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:00<00:00, 19.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:00<00:00, 19.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:01<00:00, 18.87it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:00, 17.28it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 18.65it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:00<00:00, 19.57it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 18.94it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 19.12it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 19.00it/s]
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:44 [gpu_model_runner.py:4548] Graph capturing finished in 2 secs, took 1.74 GiB
+[0;36m(EngineCore_DP0 pid=11713)[0;0m INFO 12-09 19:32:44 [core.py:256] init engine (profile, create kv cache, warmup model) took 41.14 seconds
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=11550)[0;0m WARNING 12-09 19:32:45 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [serving_responses.py:197] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:32:45 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=11550)[0;0m INFO: Started server process [11550]
+[0;36m(APIServer pid=11550)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=11550)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:42186 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:33:06 [loggers.py:248] Engine 000: Avg prompt throughput: 4.8 tokens/s, Avg generation throughput: 16.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:33:16 [loggers.py:248] Engine 000: Avg prompt throughput: 133.8 tokens/s, Avg generation throughput: 102.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:33:26 [loggers.py:248] Engine 000: Avg prompt throughput: 224.6 tokens/s, Avg generation throughput: 142.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:34506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:33:36 [loggers.py:248] Engine 000: Avg prompt throughput: 279.4 tokens/s, Avg generation throughput: 198.1 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:34506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:34506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:33:46 [loggers.py:248] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 185.0 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:33:56 [loggers.py:248] Engine 000: Avg prompt throughput: 374.5 tokens/s, Avg generation throughput: 205.5 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:48102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:34:06 [loggers.py:248] Engine 000: Avg prompt throughput: 343.4 tokens/s, Avg generation throughput: 232.9 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:34506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:34:16 [loggers.py:248] Engine 000: Avg prompt throughput: 278.0 tokens/s, Avg generation throughput: 213.0 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:34506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:34:26 [loggers.py:248] Engine 000: Avg prompt throughput: 295.0 tokens/s, Avg generation throughput: 198.9 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:34:36 [loggers.py:248] Engine 000: Avg prompt throughput: 176.4 tokens/s, Avg generation throughput: 279.2 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:34506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:34506 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:34:46 [loggers.py:248] Engine 000: Avg prompt throughput: 284.7 tokens/s, Avg generation throughput: 292.9 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:34:56 [loggers.py:248] Engine 000: Avg prompt throughput: 40.2 tokens/s, Avg generation throughput: 289.2 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:35:06 [loggers.py:248] Engine 000: Avg prompt throughput: 229.6 tokens/s, Avg generation throughput: 186.7 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:35:16 [loggers.py:248] Engine 000: Avg prompt throughput: 219.1 tokens/s, Avg generation throughput: 250.1 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:35:26 [loggers.py:248] Engine 000: Avg prompt throughput: 156.4 tokens/s, Avg generation throughput: 232.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:35:36 [loggers.py:248] Engine 000: Avg prompt throughput: 70.6 tokens/s, Avg generation throughput: 158.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:35904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:35:46 [loggers.py:248] Engine 000: Avg prompt throughput: 172.9 tokens/s, Avg generation throughput: 180.7 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:36546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:35:56 [loggers.py:248] Engine 000: Avg prompt throughput: 73.3 tokens/s, Avg generation throughput: 175.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:36:06 [loggers.py:248] Engine 000: Avg prompt throughput: 263.0 tokens/s, Avg generation throughput: 245.3 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:36:16 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 176.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:36:26 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 74.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:36:36 [loggers.py:248] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 9.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:36:46 [loggers.py:248] Engine 000: Avg prompt throughput: 516.5 tokens/s, Avg generation throughput: 155.4 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 3.8%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:36:56 [loggers.py:248] Engine 000: Avg prompt throughput: 1210.6 tokens/s, Avg generation throughput: 535.3 tokens/s, Running: 36 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.8%, Prefix cache hit rate: 18.6%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:37:06 [loggers.py:248] Engine 000: Avg prompt throughput: 909.5 tokens/s, Avg generation throughput: 760.7 tokens/s, Running: 38 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.1%, Prefix cache hit rate: 29.5%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:37:16 [loggers.py:248] Engine 000: Avg prompt throughput: 653.2 tokens/s, Avg generation throughput: 869.0 tokens/s, Running: 51 reqs, Waiting: 0 reqs, GPU KV cache usage: 25.3%, Prefix cache hit rate: 35.6%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:37:26 [loggers.py:248] Engine 000: Avg prompt throughput: 684.2 tokens/s, Avg generation throughput: 840.2 tokens/s, Running: 46 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.8%, Prefix cache hit rate: 39.1%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:37:36 [loggers.py:248] Engine 000: Avg prompt throughput: 946.5 tokens/s, Avg generation throughput: 783.9 tokens/s, Running: 50 reqs, Waiting: 0 reqs, GPU KV cache usage: 26.7%, Prefix cache hit rate: 34.9%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:37:46 [loggers.py:248] Engine 000: Avg prompt throughput: 1051.5 tokens/s, Avg generation throughput: 876.2 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 27.4%, Prefix cache hit rate: 31.2%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:37:56 [loggers.py:248] Engine 000: Avg prompt throughput: 376.9 tokens/s, Avg generation throughput: 905.8 tokens/s, Running: 50 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.3%, Prefix cache hit rate: 30.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:38:06 [loggers.py:248] Engine 000: Avg prompt throughput: 960.4 tokens/s, Avg generation throughput: 956.4 tokens/s, Running: 64 reqs, Waiting: 2 reqs, GPU KV cache usage: 22.9%, Prefix cache hit rate: 27.4%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:38:16 [loggers.py:248] Engine 000: Avg prompt throughput: 632.8 tokens/s, Avg generation throughput: 1012.1 tokens/s, Running: 60 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.4%, Prefix cache hit rate: 25.9%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:38:26 [loggers.py:248] Engine 000: Avg prompt throughput: 1205.1 tokens/s, Avg generation throughput: 825.3 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 26.2%, Prefix cache hit rate: 23.5%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:38:36 [loggers.py:248] Engine 000: Avg prompt throughput: 534.7 tokens/s, Avg generation throughput: 792.4 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 22.4%, Prefix cache hit rate: 22.6%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:38:46 [loggers.py:248] Engine 000: Avg prompt throughput: 765.4 tokens/s, Avg generation throughput: 734.2 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.8%, Prefix cache hit rate: 21.4%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:38:56 [loggers.py:248] Engine 000: Avg prompt throughput: 697.9 tokens/s, Avg generation throughput: 832.7 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.6%, Prefix cache hit rate: 20.5%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:39:06 [loggers.py:248] Engine 000: Avg prompt throughput: 877.0 tokens/s, Avg generation throughput: 706.3 tokens/s, Running: 42 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.1%, Prefix cache hit rate: 19.4%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:55000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:47596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:39:16 [loggers.py:248] Engine 000: Avg prompt throughput: 721.0 tokens/s, Avg generation throughput: 777.6 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.1%, Prefix cache hit rate: 18.6%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:39:26 [loggers.py:248] Engine 000: Avg prompt throughput: 555.0 tokens/s, Avg generation throughput: 818.7 tokens/s, Running: 50 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.5%, Prefix cache hit rate: 18.0%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:56598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:39:36 [loggers.py:248] Engine 000: Avg prompt throughput: 1128.5 tokens/s, Avg generation throughput: 789.6 tokens/s, Running: 52 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.7%, Prefix cache hit rate: 16.8%
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:45554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:37216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:49418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:39:46 [loggers.py:248] Engine 000: Avg prompt throughput: 242.3 tokens/s, Avg generation throughput: 916.4 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.9%, Prefix cache hit rate: 16.6%
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:39:56 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 444.1 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.3%, Prefix cache hit rate: 16.6%
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:40:06 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 157.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.5%, Prefix cache hit rate: 16.6%
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:40:16 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1209 19:40:16.062937543 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
+[0;36m(APIServer pid=11550)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=11550)[0;0m INFO 12-09 19:40:16 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 72.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 16.6%
+[0;36m(APIServer pid=11550)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=11550)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..38bfb5d
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 544.9470995769998,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 1.8350405035208417,
+ "tokens_per_second": 1360.3779166371198
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..edd313e
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-a3fea59e-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 196.99 \nTotal input tokens: 40429 \nTotal generated tokens: 38821 \nRequest throughput (req/s): 0.91 \nOutput token throughput (tok/s): 197.07 \nPeak output token throughput (tok/s): 340.00 \nPeak concurrent requests: 15.00 \nTotal Token throughput (tok/s): 402.31 \n---------------Time to First Token----------------\nMean TTFT (ms): 104.59 \nMedian TTFT (ms): 82.18 \nP99 TTFT (ms): 271.75 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 38.86 \nMedian TPOT (ms): 38.92 \nP99 TPOT (ms): 50.07 \n---------------Inter-token Latency----------------\nMean ITL (ms): 38.20 \nMedian ITL (ms): 32.74 \nP99 ITL (ms): 86.49 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..ee001de
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-f36abb16-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 241.29 \nTotal input tokens: 158086 \nTotal generated tokens: 152183 \nRequest throughput (req/s): 2.98 \nOutput token throughput (tok/s): 630.70 \nPeak output token throughput (tok/s): 896.00 \nPeak concurrent requests: 152.00 \nTotal Token throughput (tok/s): 1285.86 \n---------------Time to First Token----------------\nMean TTFT (ms): 12436.56 \nMedian TTFT (ms): 17115.90 \nP99 TTFT (ms): 24750.51 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 84.04 \nMedian TPOT (ms): 85.32 \nP99 TPOT (ms): 115.82 \n---------------Inter-token Latency----------------\nMean ITL (ms): 83.72 \nMedian ITL (ms): 75.54 \nP99 ITL (ms): 261.14 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..c03a57f
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1048 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:45:29 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:45:29 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=5124)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:45:33 [model.py:629] Resolved architecture: Gemma3ForConditionalGeneration
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:45:33 [model.py:1755] Using max model len 32768
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:45:33 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:38 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='RedHatAI/gemma-3-12b-it-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/gemma-3-12b-it-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/gemma-3-12b-it-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:40 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:45401 backend=nccl
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:40 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=5288)[0;0m Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:45 [gpu_model_runner.py:3544] Starting to load model RedHatAI/gemma-3-12b-it-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=5288)[0;0m WARNING 12-11 18:45:46 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:46 [layer.py:524] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(EngineCore_DP0 pid=5288)[0;0m WARNING 12-11 18:45:46 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:46 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=5288)[0;0m WARNING 12-11 18:45:46 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:01<00:03, 1.57s/it]
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:02<00:01, 1.36s/it]
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:03<00:00, 1.15s/it]
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:03<00:00, 1.23s/it]
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:50 [default_loader.py:308] Loading weights took 3.74 seconds
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:50 [gpu_model_runner.py:3626] Model loading took 13.5273 GiB memory and 4.353595 seconds
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:51 [gpu_model_runner.py:4388] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:58 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/55ce54d8f2/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:45:58 [backends.py:676] Dynamo bytecode transform time: 6.40 s
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:02 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:06 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 4.76 s
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:06 [monitor.py:34] torch.compile takes 11.16 s in total
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:09 [gpu_worker.py:364] Available KV cache memory: 16.25 GiB
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:09 [kv_cache_utils.py:1287] GPU KV cache size: 44,368 tokens
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:09 [kv_cache_utils.py:1292] Maximum concurrency for 32,768 tokens per request: 5.52x
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 19.11it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 21.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 22.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 23.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 24.20it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 24.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 23.98it/s]
+[0;36m(EngineCore_DP0 pid=5288)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 21.95it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 24.62it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 24.80it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 24.71it/s]
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:11 [gpu_model_runner.py:4548] Graph capturing finished in 2 secs, took 1.38 GiB
+[0;36m(EngineCore_DP0 pid=5288)[0;0m INFO 12-11 18:46:11 [core.py:256] init engine (profile, create kv cache, warmup model) took 20.95 seconds
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:13 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:14 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=5124)[0;0m INFO: Started server process [5124]
+[0;36m(APIServer pid=5124)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=5124)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:46170 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:34 [loggers.py:248] Engine 000: Avg prompt throughput: 42.1 tokens/s, Avg generation throughput: 39.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:44 [loggers.py:248] Engine 000: Avg prompt throughput: 138.6 tokens/s, Avg generation throughput: 117.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:46:54 [loggers.py:248] Engine 000: Avg prompt throughput: 205.5 tokens/s, Avg generation throughput: 199.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:47:04 [loggers.py:248] Engine 000: Avg prompt throughput: 297.3 tokens/s, Avg generation throughput: 183.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:47:14 [loggers.py:248] Engine 000: Avg prompt throughput: 202.2 tokens/s, Avg generation throughput: 184.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:47:24 [loggers.py:248] Engine 000: Avg prompt throughput: 397.0 tokens/s, Avg generation throughput: 183.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:47:34 [loggers.py:248] Engine 000: Avg prompt throughput: 380.0 tokens/s, Avg generation throughput: 204.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:50734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:47:44 [loggers.py:248] Engine 000: Avg prompt throughput: 170.5 tokens/s, Avg generation throughput: 216.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:47:54 [loggers.py:248] Engine 000: Avg prompt throughput: 397.2 tokens/s, Avg generation throughput: 183.9 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:55596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:55612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:48:04 [loggers.py:248] Engine 000: Avg prompt throughput: 230.9 tokens/s, Avg generation throughput: 256.9 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:48:14 [loggers.py:248] Engine 000: Avg prompt throughput: 292.5 tokens/s, Avg generation throughput: 284.6 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:48:24 [loggers.py:248] Engine 000: Avg prompt throughput: 38.6 tokens/s, Avg generation throughput: 252.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:55596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:48:34 [loggers.py:248] Engine 000: Avg prompt throughput: 265.7 tokens/s, Avg generation throughput: 198.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:55596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:48:44 [loggers.py:248] Engine 000: Avg prompt throughput: 290.7 tokens/s, Avg generation throughput: 235.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:40096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:48:54 [loggers.py:248] Engine 000: Avg prompt throughput: 106.1 tokens/s, Avg generation throughput: 217.9 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:49:04 [loggers.py:248] Engine 000: Avg prompt throughput: 109.9 tokens/s, Avg generation throughput: 195.2 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:49:14 [loggers.py:248] Engine 000: Avg prompt throughput: 110.6 tokens/s, Avg generation throughput: 176.9 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:43116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:49:24 [loggers.py:248] Engine 000: Avg prompt throughput: 242.3 tokens/s, Avg generation throughput: 192.2 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:45752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:41348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:49574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:49:34 [loggers.py:248] Engine 000: Avg prompt throughput: 126.3 tokens/s, Avg generation throughput: 228.8 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:49:44 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:49:54 [loggers.py:248] Engine 000: Avg prompt throughput: 1.1 tokens/s, Avg generation throughput: 17.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:50:04 [loggers.py:248] Engine 000: Avg prompt throughput: 459.6 tokens/s, Avg generation throughput: 178.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:50:14 [loggers.py:248] Engine 000: Avg prompt throughput: 1338.0 tokens/s, Avg generation throughput: 420.2 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 25.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:50:24 [loggers.py:248] Engine 000: Avg prompt throughput: 975.7 tokens/s, Avg generation throughput: 601.7 tokens/s, Running: 46 reqs, Waiting: 0 reqs, GPU KV cache usage: 33.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:50:34 [loggers.py:248] Engine 000: Avg prompt throughput: 681.3 tokens/s, Avg generation throughput: 656.1 tokens/s, Running: 51 reqs, Waiting: 0 reqs, GPU KV cache usage: 39.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:50:44 [loggers.py:248] Engine 000: Avg prompt throughput: 684.8 tokens/s, Avg generation throughput: 709.3 tokens/s, Running: 62 reqs, Waiting: 5 reqs, GPU KV cache usage: 51.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:50:54 [loggers.py:248] Engine 000: Avg prompt throughput: 565.3 tokens/s, Avg generation throughput: 697.6 tokens/s, Running: 63 reqs, Waiting: 19 reqs, GPU KV cache usage: 56.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:51:04 [loggers.py:248] Engine 000: Avg prompt throughput: 1034.7 tokens/s, Avg generation throughput: 620.8 tokens/s, Running: 62 reqs, Waiting: 24 reqs, GPU KV cache usage: 55.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:51:14 [loggers.py:248] Engine 000: Avg prompt throughput: 503.9 tokens/s, Avg generation throughput: 691.2 tokens/s, Running: 64 reqs, Waiting: 38 reqs, GPU KV cache usage: 60.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:51:24 [loggers.py:248] Engine 000: Avg prompt throughput: 654.6 tokens/s, Avg generation throughput: 684.7 tokens/s, Running: 64 reqs, Waiting: 54 reqs, GPU KV cache usage: 53.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:51:34 [loggers.py:248] Engine 000: Avg prompt throughput: 390.7 tokens/s, Avg generation throughput: 723.2 tokens/s, Running: 62 reqs, Waiting: 67 reqs, GPU KV cache usage: 47.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:51:44 [loggers.py:248] Engine 000: Avg prompt throughput: 608.1 tokens/s, Avg generation throughput: 755.2 tokens/s, Running: 63 reqs, Waiting: 78 reqs, GPU KV cache usage: 44.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:51:54 [loggers.py:248] Engine 000: Avg prompt throughput: 640.7 tokens/s, Avg generation throughput: 768.0 tokens/s, Running: 63 reqs, Waiting: 75 reqs, GPU KV cache usage: 46.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:52:04 [loggers.py:248] Engine 000: Avg prompt throughput: 820.3 tokens/s, Avg generation throughput: 748.8 tokens/s, Running: 63 reqs, Waiting: 72 reqs, GPU KV cache usage: 46.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:52:14 [loggers.py:248] Engine 000: Avg prompt throughput: 1096.4 tokens/s, Avg generation throughput: 678.4 tokens/s, Running: 64 reqs, Waiting: 75 reqs, GPU KV cache usage: 47.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:52:24 [loggers.py:248] Engine 000: Avg prompt throughput: 752.2 tokens/s, Avg generation throughput: 742.4 tokens/s, Running: 63 reqs, Waiting: 74 reqs, GPU KV cache usage: 49.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:52:34 [loggers.py:248] Engine 000: Avg prompt throughput: 659.8 tokens/s, Avg generation throughput: 761.5 tokens/s, Running: 64 reqs, Waiting: 70 reqs, GPU KV cache usage: 46.1%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:52:44 [loggers.py:248] Engine 000: Avg prompt throughput: 1211.2 tokens/s, Avg generation throughput: 691.2 tokens/s, Running: 64 reqs, Waiting: 62 reqs, GPU KV cache usage: 45.1%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:34798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:36726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:58176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:59598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:60608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:54706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:52:54 [loggers.py:248] Engine 000: Avg prompt throughput: 749.0 tokens/s, Avg generation throughput: 768.0 tokens/s, Running: 63 reqs, Waiting: 78 reqs, GPU KV cache usage: 47.7%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:38474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:38484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:47504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:38492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:38494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:38498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:38508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:53:04 [loggers.py:248] Engine 000: Avg prompt throughput: 435.1 tokens/s, Avg generation throughput: 800.0 tokens/s, Running: 64 reqs, Waiting: 67 reqs, GPU KV cache usage: 48.5%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:53:14 [loggers.py:248] Engine 000: Avg prompt throughput: 762.0 tokens/s, Avg generation throughput: 723.2 tokens/s, Running: 63 reqs, Waiting: 35 reqs, GPU KV cache usage: 51.3%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:53:24 [loggers.py:248] Engine 000: Avg prompt throughput: 784.6 tokens/s, Avg generation throughput: 701.9 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 46.7%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:53:34 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 640.0 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.2%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:53:44 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 350.8 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.6%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:53:54 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 101.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=5124)[0;0m INFO 12-11 18:53:57 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1211 18:53:58.272992418 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
+[0;36m(APIServer pid=5124)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=5124)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=5124)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..e6da6af
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 878.6614384089999,
+ "num_requests": 1000,
+ "total_num_tokens": 755432,
+ "requests_per_second": 1.1380947840509639,
+ "tokens_per_second": 859.7532189051877
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..482a5a1
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-a1348ed0-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 190.66 \nTotal input tokens: 40429 \nTotal generated tokens: 39127 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 205.22 \nPeak output token throughput (tok/s): 422.00 \nPeak concurrent requests: 11.00 \nTotal Token throughput (tok/s): 417.26 \n---------------Time to First Token----------------\nMean TTFT (ms): 89.11 \nMedian TTFT (ms): 64.29 \nP99 TTFT (ms): 255.75 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 22.00 \nMedian TPOT (ms): 21.60 \nP99 TPOT (ms): 28.81 \n---------------Inter-token Latency----------------\nMean ITL (ms): 21.91 \nMedian ITL (ms): 20.88 \nP99 ITL (ms): 45.88 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..fd03801
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-12b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-370f24dd-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 203.05 \nTotal input tokens: 158086 \nTotal generated tokens: 153200 \nRequest throughput (req/s): 3.55 \nOutput token throughput (tok/s): 754.51 \nPeak output token throughput (tok/s): 1152.00 \nPeak concurrent requests: 78.00 \nTotal Token throughput (tok/s): 1533.08 \n---------------Time to First Token----------------\nMean TTFT (ms): 222.29 \nMedian TTFT (ms): 103.65 \nP99 TTFT (ms): 2503.58 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 54.74 \nMedian TPOT (ms): 55.60 \nP99 TPOT (ms): 83.43 \n---------------Inter-token Latency----------------\nMean ITL (ms): 54.25 \nMedian ITL (ms): 49.31 \nP99 ITL (ms): 204.91 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log
new file mode 100644
index 0000000..1afb9aa
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_server.log
@@ -0,0 +1,1074 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:27:42 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:27:42 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/gemma-3-12b-it-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 9900, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=10267)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:27:46 [model.py:629] Resolved architecture: Gemma3ForConditionalGeneration
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:27:46 [model.py:1755] Using max model len 9900
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:27:46 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=10431)[0;0m INFO 12-11 19:27:51 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='RedHatAI/gemma-3-12b-it-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/gemma-3-12b-it-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=9900, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/gemma-3-12b-it-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=10431)[0;0m WARNING 12-11 19:27:51 [multiproc_executor.py:880] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+INFO 12-11 19:27:56 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:56779 backend=nccl
+INFO 12-11 19:27:56 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:56779 backend=nccl
+INFO 12-11 19:27:56 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-11 19:27:57 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-11 19:27:57 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:03 [gpu_model_runner.py:3544] Starting to load model RedHatAI/gemma-3-12b-it-FP8-dynamic...
+[0;36m(Worker_TP1 pid=10514)[0;0m WARNING 12-11 19:28:03 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP1 pid=10514)[0;0m INFO 12-11 19:28:03 [layer.py:524] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP1 pid=10514)[0;0m WARNING 12-11 19:28:03 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP1 pid=10514)[0;0m INFO 12-11 19:28:03 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=10514)[0;0m WARNING 12-11 19:28:03 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP0 pid=10513)[0;0m WARNING 12-11 19:28:03 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:03 [layer.py:524] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP0 pid=10513)[0;0m WARNING 12-11 19:28:03 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:03 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=10513)[0;0m WARNING 12-11 19:28:03 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP0 pid=10513)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=10513)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:01<00:02, 1.19s/it]
+[0;36m(Worker_TP0 pid=10513)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:02<00:01, 1.06s/it]
+[0;36m(Worker_TP0 pid=10513)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.11it/s]
+[0;36m(Worker_TP0 pid=10513)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.05it/s]
+[0;36m(Worker_TP0 pid=10513)[0;0m
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:07 [default_loader.py:308] Loading weights took 2.89 seconds
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:07 [gpu_model_runner.py:3626] Model loading took 7.1152 GiB memory and 3.610627 seconds
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:07 [gpu_model_runner.py:4388] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP1 pid=10514)[0;0m INFO 12-11 19:28:07 [gpu_model_runner.py:4388] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:16 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/8a51c8e3ac/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:16 [backends.py:676] Dynamo bytecode transform time: 7.15 s
+[0;36m(Worker_TP1 pid=10514)[0;0m INFO 12-11 19:28:19 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:19 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:36 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 17.45 s
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:36 [monitor.py:34] torch.compile takes 24.59 s in total
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:40 [gpu_worker.py:364] Available KV cache memory: 22.78 GiB
+[0;36m(EngineCore_DP0 pid=10431)[0;0m INFO 12-11 19:28:40 [kv_cache_utils.py:1287] GPU KV cache size: 124,416 tokens
+[0;36m(EngineCore_DP0 pid=10431)[0;0m INFO 12-11 19:28:40 [kv_cache_utils.py:1292] Maximum concurrency for 9,900 tokens per request: 29.46x
+[0;36m(Worker_TP0 pid=10513)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:07, 2.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:06, 2.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:01<00:06, 2.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:01<00:05, 2.51it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:01<00:05, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:02<00:05, 2.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:02<00:04, 2.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:03<00:04, 2.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:03<00:04, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:04<00:03, 2.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:04<00:03, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:04<00:02, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:05<00:02, 2.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:05<00:01, 2.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:06<00:01, 2.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:06<00:01, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:06<00:00, 2.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:07<00:00, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.50it/s]
+[0;36m(Worker_TP0 pid=10513)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:01<00:11, 1.12s/it]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:02<00:10, 1.14s/it]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:02<00:06, 1.27it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:03<00:04, 1.58it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:03<00:03, 1.84it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:04<00:03, 1.47it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:05<00:03, 1.30it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:05<00:01, 1.56it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:06<00:01, 1.80it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:06<00:00, 2.01it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:07<00:00, 1.58it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:07<00:00, 1.49it/s]
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:28:56 [gpu_model_runner.py:4548] Graph capturing finished in 16 secs, took 0.98 GiB
+[0;36m(EngineCore_DP0 pid=10431)[0;0m INFO 12-11 19:28:56 [core.py:256] init engine (profile, create kv cache, warmup model) took 48.99 seconds
+[0;36m(EngineCore_DP0 pid=10431)[0;0m Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:04 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:05 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=10267)[0;0m INFO: Started server process [10267]
+[0;36m(APIServer pid=10267)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=10267)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:54092 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:25 [loggers.py:248] Engine 000: Avg prompt throughput: 45.0 tokens/s, Avg generation throughput: 69.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:35 [loggers.py:248] Engine 000: Avg prompt throughput: 178.1 tokens/s, Avg generation throughput: 146.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:45 [loggers.py:248] Engine 000: Avg prompt throughput: 163.1 tokens/s, Avg generation throughput: 239.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:29:55 [loggers.py:248] Engine 000: Avg prompt throughput: 298.2 tokens/s, Avg generation throughput: 131.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:30:05 [loggers.py:248] Engine 000: Avg prompt throughput: 231.5 tokens/s, Avg generation throughput: 199.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:30:15 [loggers.py:248] Engine 000: Avg prompt throughput: 366.8 tokens/s, Avg generation throughput: 222.1 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:30:25 [loggers.py:248] Engine 000: Avg prompt throughput: 399.0 tokens/s, Avg generation throughput: 232.9 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:46922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:30:35 [loggers.py:248] Engine 000: Avg prompt throughput: 153.9 tokens/s, Avg generation throughput: 178.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:34954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:34954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:34962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:30:45 [loggers.py:248] Engine 000: Avg prompt throughput: 417.8 tokens/s, Avg generation throughput: 239.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:34954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:34962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:30:55 [loggers.py:248] Engine 000: Avg prompt throughput: 207.9 tokens/s, Avg generation throughput: 354.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:59418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:34954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:48848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:34954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:31:05 [loggers.py:248] Engine 000: Avg prompt throughput: 292.5 tokens/s, Avg generation throughput: 203.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:31:15 [loggers.py:248] Engine 000: Avg prompt throughput: 45.8 tokens/s, Avg generation throughput: 199.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:31:25 [loggers.py:248] Engine 000: Avg prompt throughput: 305.6 tokens/s, Avg generation throughput: 211.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:31:35 [loggers.py:248] Engine 000: Avg prompt throughput: 247.2 tokens/s, Avg generation throughput: 255.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:56662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:51832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:31:45 [loggers.py:248] Engine 000: Avg prompt throughput: 103.5 tokens/s, Avg generation throughput: 261.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:56662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:31:55 [loggers.py:248] Engine 000: Avg prompt throughput: 145.2 tokens/s, Avg generation throughput: 96.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:56662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:32:05 [loggers.py:248] Engine 000: Avg prompt throughput: 90.8 tokens/s, Avg generation throughput: 192.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:60876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:35480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:35490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:32:15 [loggers.py:248] Engine 000: Avg prompt throughput: 233.8 tokens/s, Avg generation throughput: 209.8 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:35490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:56662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:35480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:32:25 [loggers.py:248] Engine 000: Avg prompt throughput: 118.3 tokens/s, Avg generation throughput: 243.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:32:35 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 38.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:32:45 [loggers.py:248] Engine 000: Avg prompt throughput: 386.2 tokens/s, Avg generation throughput: 205.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 8.2%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:32:55 [loggers.py:248] Engine 000: Avg prompt throughput: 1295.5 tokens/s, Avg generation throughput: 643.6 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.4%, Prefix cache hit rate: 28.2%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:33:05 [loggers.py:248] Engine 000: Avg prompt throughput: 1072.0 tokens/s, Avg generation throughput: 692.3 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.9%, Prefix cache hit rate: 39.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:33:15 [loggers.py:248] Engine 000: Avg prompt throughput: 701.1 tokens/s, Avg generation throughput: 784.5 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.7%, Prefix cache hit rate: 44.4%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:33:25 [loggers.py:248] Engine 000: Avg prompt throughput: 671.8 tokens/s, Avg generation throughput: 832.7 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.0%, Prefix cache hit rate: 47.6%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:33:35 [loggers.py:248] Engine 000: Avg prompt throughput: 1031.9 tokens/s, Avg generation throughput: 719.6 tokens/s, Running: 46 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.1%, Prefix cache hit rate: 42.2%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:33:45 [loggers.py:248] Engine 000: Avg prompt throughput: 1086.0 tokens/s, Avg generation throughput: 803.4 tokens/s, Running: 56 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.0%, Prefix cache hit rate: 37.8%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:33:55 [loggers.py:248] Engine 000: Avg prompt throughput: 509.4 tokens/s, Avg generation throughput: 883.0 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.0%, Prefix cache hit rate: 36.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:34:05 [loggers.py:248] Engine 000: Avg prompt throughput: 911.4 tokens/s, Avg generation throughput: 927.9 tokens/s, Running: 64 reqs, Waiting: 10 reqs, GPU KV cache usage: 16.5%, Prefix cache hit rate: 33.2%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:34:15 [loggers.py:248] Engine 000: Avg prompt throughput: 684.7 tokens/s, Avg generation throughput: 1015.7 tokens/s, Running: 58 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.9%, Prefix cache hit rate: 31.4%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:34:25 [loggers.py:248] Engine 000: Avg prompt throughput: 952.5 tokens/s, Avg generation throughput: 900.2 tokens/s, Running: 57 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.8%, Prefix cache hit rate: 29.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:42508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:34:35 [loggers.py:248] Engine 000: Avg prompt throughput: 1029.1 tokens/s, Avg generation throughput: 754.8 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.1%, Prefix cache hit rate: 27.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:34:45 [loggers.py:248] Engine 000: Avg prompt throughput: 476.6 tokens/s, Avg generation throughput: 809.6 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.0%, Prefix cache hit rate: 26.2%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:34:55 [loggers.py:248] Engine 000: Avg prompt throughput: 980.0 tokens/s, Avg generation throughput: 842.8 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.4%, Prefix cache hit rate: 24.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:35:05 [loggers.py:248] Engine 000: Avg prompt throughput: 944.0 tokens/s, Avg generation throughput: 757.0 tokens/s, Running: 29 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.6%, Prefix cache hit rate: 23.3%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:35:15 [loggers.py:248] Engine 000: Avg prompt throughput: 776.3 tokens/s, Avg generation throughput: 699.0 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.8%, Prefix cache hit rate: 22.3%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:35:25 [loggers.py:248] Engine 000: Avg prompt throughput: 640.6 tokens/s, Avg generation throughput: 754.1 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.4%, Prefix cache hit rate: 21.5%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:35:35 [loggers.py:248] Engine 000: Avg prompt throughput: 1027.1 tokens/s, Avg generation throughput: 796.5 tokens/s, Running: 49 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.3%, Prefix cache hit rate: 20.3%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:45408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:37992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:58596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:38030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:33210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:36890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:44618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO: 127.0.0.1:52316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:35:45 [loggers.py:248] Engine 000: Avg prompt throughput: 632.9 tokens/s, Avg generation throughput: 857.1 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.9%, Prefix cache hit rate: 19.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:35:55 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 592.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 19.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:36:02 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP1 pid=10514)[0;0m INFO 12-11 19:36:02 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(Worker_TP0 pid=10513)[0;0m INFO 12-11 19:36:02 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(APIServer pid=10267)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=10267)[0;0m INFO 12-11 19:36:05 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 60.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 19.7%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=10267)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=10267)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json
new file mode 100644
index 0000000..b3e37f7
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-12b-it-FP8-dynamic_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 569.8341109329999,
+ "num_requests": 1000,
+ "total_num_tokens": 755432,
+ "requests_per_second": 1.7548966985543941,
+ "tokens_per_second": 1325.705122782343
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..86f7a26
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-27b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-eb2c00d0-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 202.25 \nTotal input tokens: 40429 \nTotal generated tokens: 39398 \nRequest throughput (req/s): 0.89 \nOutput token throughput (tok/s): 194.80 \nPeak output token throughput (tok/s): 308.00 \nPeak concurrent requests: 17.00 \nTotal Token throughput (tok/s): 394.70 \n---------------Time to First Token----------------\nMean TTFT (ms): 135.44 \nMedian TTFT (ms): 102.54 \nP99 TTFT (ms): 390.74 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 47.24 \nMedian TPOT (ms): 46.60 \nP99 TPOT (ms): 56.08 \n---------------Inter-token Latency----------------\nMean ITL (ms): 47.00 \nMedian ITL (ms): 45.58 \nP99 ITL (ms): 140.61 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..1c37ec0
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/gemma-3-27b-it-FP8-dynamic', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-eb0dc17a-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 320.24 \nTotal input tokens: 158086 \nTotal generated tokens: 152563 \nRequest throughput (req/s): 2.25 \nOutput token throughput (tok/s): 476.40 \nPeak output token throughput (tok/s): 608.00 \nPeak concurrent requests: 304.00 \nTotal Token throughput (tok/s): 970.05 \n---------------Time to First Token----------------\nMean TTFT (ms): 52928.91 \nMedian TTFT (ms): 59992.28 \nP99 TTFT (ms): 104765.35 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 61.40 \nMedian TPOT (ms): 60.27 \nP99 TPOT (ms): 90.17 \n---------------Inter-token Latency----------------\nMean ITL (ms): 60.58 \nMedian ITL (ms): 53.74 \nP99 ITL (ms): 244.74 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log
new file mode 100644
index 0000000..67be5ec
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_server.log
@@ -0,0 +1,1092 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:47:57 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:47:57 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/gemma-3-27b-it-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/gemma-3-27b-it-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 29000, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.94, 'max_num_seqs': 32}
+[0;36m(APIServer pid=12756)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:48:02 [model.py:629] Resolved architecture: Gemma3ForConditionalGeneration
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:48:02 [model.py:1755] Using max model len 29000
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:48:02 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=12920)[0;0m INFO 12-10 16:48:07 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='RedHatAI/gemma-3-27b-it-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/gemma-3-27b-it-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=29000, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=RedHatAI/gemma-3-27b-it-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 64, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=12920)[0;0m WARNING 12-10 16:48:07 [multiproc_executor.py:880] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+INFO 12-10 16:48:11 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:41237 backend=nccl
+INFO 12-10 16:48:11 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:41237 backend=nccl
+INFO 12-10 16:48:11 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-10 16:48:12 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-10 16:48:12 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:18 [gpu_model_runner.py:3544] Starting to load model RedHatAI/gemma-3-27b-it-FP8-dynamic...
+[0;36m(Worker_TP1 pid=13003)[0;0m WARNING 12-10 16:48:18 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP1 pid=13003)[0;0m INFO 12-10 16:48:18 [layer.py:524] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP1 pid=13003)[0;0m WARNING 12-10 16:48:18 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP1 pid=13003)[0;0m INFO 12-10 16:48:18 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=13003)[0;0m WARNING 12-10 16:48:18 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP0 pid=13002)[0;0m WARNING 12-10 16:48:18 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:18 [layer.py:524] Using AttentionBackendEnum.TORCH_SDPA for MultiHeadAttention in multimodal encoder.
+[0;36m(Worker_TP0 pid=13002)[0;0m WARNING 12-10 16:48:18 [activation.py:544] [ROCm] PyTorch's native GELU with tanh approximation is unstable. Falling back to GELU(approximate='none').
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:18 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=13002)[0;0m WARNING 12-10 16:48:18 [activation.py:220] [ROCm] PyTorch's native GELU with tanh approximation is unstable with torch.compile. For native implementation, fallback to 'none' approximation. The custom kernel implementation is unaffected.
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/6 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 17% Completed | 1/6 [00:00<00:03, 1.35it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 2/6 [00:01<00:02, 1.34it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 3/6 [00:02<00:02, 1.33it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 4/6 [00:02<00:01, 1.40it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 83% Completed | 5/6 [00:03<00:00, 1.43it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 6/6 [00:04<00:00, 1.23it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 6/6 [00:04<00:00, 1.30it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:23 [default_loader.py:308] Loading weights took 4.63 seconds
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:24 [gpu_model_runner.py:3626] Model loading took 14.2812 GiB memory and 5.335814 seconds
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:24 [gpu_model_runner.py:4388] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP1 pid=13003)[0;0m INFO 12-10 16:48:24 [gpu_model_runner.py:4388] Encoder cache will be initialized with a budget of 2048 tokens, and profiled with 7 image items of the maximum feature size.
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:35 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/ea06340fec/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:35 [backends.py:676] Dynamo bytecode transform time: 9.05 s
+[0;36m(Worker_TP1 pid=13003)[0;0m INFO 12-10 16:48:39 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:48:39 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:49:08 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 28.71 s
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:49:08 [monitor.py:34] torch.compile takes 37.76 s in total
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:49:11 [gpu_worker.py:364] Available KV cache memory: 14.71 GiB
+[0;36m(EngineCore_DP0 pid=12920)[0;0m WARNING 12-10 16:49:12 [kv_cache_utils.py:1029] Add 8 padding layers, may waste at most 15.38% KV cache memory
+[0;36m(EngineCore_DP0 pid=12920)[0;0m INFO 12-10 16:49:12 [kv_cache_utils.py:1287] GPU KV cache size: 54,960 tokens
+[0;36m(EngineCore_DP0 pid=12920)[0;0m INFO 12-10 16:49:12 [kv_cache_utils.py:1292] Maximum concurrency for 29,000 tokens per request: 8.09x
+[0;36m(Worker_TP0 pid=13002)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 9%|▉ | 1/11 [00:00<00:05, 1.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 2/11 [00:01<00:04, 1.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 3/11 [00:01<00:04, 1.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▋ | 4/11 [00:02<00:03, 1.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▌ | 5/11 [00:02<00:03, 1.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▍ | 6/11 [00:03<00:02, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▎ | 7/11 [00:03<00:02, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 8/11 [00:04<00:01, 1.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 9/11 [00:04<00:01, 1.94it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 91%|█████████ | 10/11 [00:05<00:00, 1.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 1.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 1.92it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 14%|█▍ | 1/7 [00:00<00:03, 1.94it/s]
Capturing CUDA graphs (decode, FULL): 29%|██▊ | 2/7 [00:01<00:02, 1.94it/s]
Capturing CUDA graphs (decode, FULL): 43%|████▎ | 3/7 [00:01<00:02, 1.95it/s]
Capturing CUDA graphs (decode, FULL): 57%|█████▋ | 4/7 [00:02<00:01, 1.94it/s]
Capturing CUDA graphs (decode, FULL): 71%|███████▏ | 5/7 [00:02<00:01, 1.97it/s]
Capturing CUDA graphs (decode, FULL): 86%|████████▌ | 6/7 [00:03<00:00, 1.97it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:03<00:00, 1.96it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:03<00:00, 1.96it/s]
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:49:22 [gpu_model_runner.py:4548] Graph capturing finished in 10 secs, took 1.60 GiB
+[0;36m(EngineCore_DP0 pid=12920)[0;0m INFO 12-10 16:49:22 [core.py:256] init engine (profile, create kv cache, warmup model) took 58.36 seconds
+[0;36m(EngineCore_DP0 pid=12920)[0;0m Using a slow image processor as `use_fast` is unset and a slow processor was saved with this model. `use_fast=True` will be the default behavior in v4.52, even if the model was saved with a slow processor. This will result in minor differences in outputs. You'll still be able to use a slow processor with `use_fast=False`.
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:31 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=12756)[0;0m INFO: Started server process [12756]
+[0;36m(APIServer pid=12756)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=12756)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58992 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:49:51 [loggers.py:248] Engine 000: Avg prompt throughput: 4.9 tokens/s, Avg generation throughput: 19.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:50:01 [loggers.py:248] Engine 000: Avg prompt throughput: 140.6 tokens/s, Avg generation throughput: 107.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:50:11 [loggers.py:248] Engine 000: Avg prompt throughput: 196.2 tokens/s, Avg generation throughput: 142.0 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:50:21 [loggers.py:248] Engine 000: Avg prompt throughput: 341.8 tokens/s, Avg generation throughput: 189.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:50:31 [loggers.py:248] Engine 000: Avg prompt throughput: 202.2 tokens/s, Avg generation throughput: 185.4 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:50:41 [loggers.py:248] Engine 000: Avg prompt throughput: 396.4 tokens/s, Avg generation throughput: 176.9 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.5%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:50:51 [loggers.py:248] Engine 000: Avg prompt throughput: 302.4 tokens/s, Avg generation throughput: 243.7 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:51:01 [loggers.py:248] Engine 000: Avg prompt throughput: 248.7 tokens/s, Avg generation throughput: 243.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:51:11 [loggers.py:248] Engine 000: Avg prompt throughput: 397.2 tokens/s, Avg generation throughput: 186.2 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:34044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:34044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:34044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:51:21 [loggers.py:248] Engine 000: Avg prompt throughput: 172.8 tokens/s, Avg generation throughput: 246.7 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:51:31 [loggers.py:248] Engine 000: Avg prompt throughput: 347.7 tokens/s, Avg generation throughput: 267.4 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:51510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:51:41 [loggers.py:248] Engine 000: Avg prompt throughput: 41.5 tokens/s, Avg generation throughput: 285.6 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:34044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:51:51 [loggers.py:248] Engine 000: Avg prompt throughput: 200.8 tokens/s, Avg generation throughput: 185.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:52:01 [loggers.py:248] Engine 000: Avg prompt throughput: 355.6 tokens/s, Avg generation throughput: 247.3 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:52:11 [loggers.py:248] Engine 000: Avg prompt throughput: 70.2 tokens/s, Avg generation throughput: 223.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:52:21 [loggers.py:248] Engine 000: Avg prompt throughput: 107.3 tokens/s, Avg generation throughput: 184.0 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:43292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:50434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:50450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:52:31 [loggers.py:248] Engine 000: Avg prompt throughput: 148.1 tokens/s, Avg generation throughput: 183.1 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:50450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:50450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:52:41 [loggers.py:248] Engine 000: Avg prompt throughput: 146.6 tokens/s, Avg generation throughput: 172.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:50450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:39644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:39644 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:39656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:39012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:39012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:39026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:52:51 [loggers.py:248] Engine 000: Avg prompt throughput: 223.0 tokens/s, Avg generation throughput: 238.7 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:53:01 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 173.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:53:11 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 50.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:49734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:53:21 [loggers.py:248] Engine 000: Avg prompt throughput: 1.1 tokens/s, Avg generation throughput: 11.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:49734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:49734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:53:31 [loggers.py:248] Engine 000: Avg prompt throughput: 698.0 tokens/s, Avg generation throughput: 201.6 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:53:41 [loggers.py:248] Engine 000: Avg prompt throughput: 963.5 tokens/s, Avg generation throughput: 436.1 tokens/s, Running: 32 reqs, Waiting: 15 reqs, GPU KV cache usage: 19.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:53:51 [loggers.py:248] Engine 000: Avg prompt throughput: 648.0 tokens/s, Avg generation throughput: 502.4 tokens/s, Running: 31 reqs, Waiting: 26 reqs, GPU KV cache usage: 21.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:54:01 [loggers.py:248] Engine 000: Avg prompt throughput: 463.9 tokens/s, Avg generation throughput: 524.8 tokens/s, Running: 31 reqs, Waiting: 40 reqs, GPU KV cache usage: 23.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:49734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:54:11 [loggers.py:248] Engine 000: Avg prompt throughput: 369.6 tokens/s, Avg generation throughput: 537.6 tokens/s, Running: 32 reqs, Waiting: 71 reqs, GPU KV cache usage: 25.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:54:21 [loggers.py:248] Engine 000: Avg prompt throughput: 503.2 tokens/s, Avg generation throughput: 521.6 tokens/s, Running: 32 reqs, Waiting: 87 reqs, GPU KV cache usage: 18.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:54:31 [loggers.py:248] Engine 000: Avg prompt throughput: 479.3 tokens/s, Avg generation throughput: 531.2 tokens/s, Running: 32 reqs, Waiting: 93 reqs, GPU KV cache usage: 20.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:49734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:54:41 [loggers.py:248] Engine 000: Avg prompt throughput: 555.7 tokens/s, Avg generation throughput: 508.8 tokens/s, Running: 32 reqs, Waiting: 111 reqs, GPU KV cache usage: 23.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:54:51 [loggers.py:248] Engine 000: Avg prompt throughput: 395.0 tokens/s, Avg generation throughput: 524.8 tokens/s, Running: 31 reqs, Waiting: 143 reqs, GPU KV cache usage: 25.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:55:01 [loggers.py:248] Engine 000: Avg prompt throughput: 657.1 tokens/s, Avg generation throughput: 492.8 tokens/s, Running: 32 reqs, Waiting: 155 reqs, GPU KV cache usage: 23.6%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:55:11 [loggers.py:248] Engine 000: Avg prompt throughput: 746.3 tokens/s, Avg generation throughput: 496.0 tokens/s, Running: 31 reqs, Waiting: 169 reqs, GPU KV cache usage: 18.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:35664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:55:21 [loggers.py:248] Engine 000: Avg prompt throughput: 154.6 tokens/s, Avg generation throughput: 579.2 tokens/s, Running: 32 reqs, Waiting: 185 reqs, GPU KV cache usage: 23.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:55:31 [loggers.py:248] Engine 000: Avg prompt throughput: 405.0 tokens/s, Avg generation throughput: 540.8 tokens/s, Running: 32 reqs, Waiting: 199 reqs, GPU KV cache usage: 20.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:41728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:55:41 [loggers.py:248] Engine 000: Avg prompt throughput: 342.2 tokens/s, Avg generation throughput: 553.7 tokens/s, Running: 31 reqs, Waiting: 226 reqs, GPU KV cache usage: 18.1%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:49734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:55:51 [loggers.py:248] Engine 000: Avg prompt throughput: 601.6 tokens/s, Avg generation throughput: 511.9 tokens/s, Running: 31 reqs, Waiting: 224 reqs, GPU KV cache usage: 16.4%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:40346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:56:01 [loggers.py:248] Engine 000: Avg prompt throughput: 334.7 tokens/s, Avg generation throughput: 556.8 tokens/s, Running: 32 reqs, Waiting: 238 reqs, GPU KV cache usage: 20.7%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:47526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:46406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:56:11 [loggers.py:248] Engine 000: Avg prompt throughput: 741.3 tokens/s, Avg generation throughput: 492.8 tokens/s, Running: 32 reqs, Waiting: 241 reqs, GPU KV cache usage: 19.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:59388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:55064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:54930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:36170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:44500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:60856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:58160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:56:21 [loggers.py:248] Engine 000: Avg prompt throughput: 966.7 tokens/s, Avg generation throughput: 454.4 tokens/s, Running: 32 reqs, Waiting: 270 reqs, GPU KV cache usage: 24.3%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: 127.0.0.1:37916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:56:31 [loggers.py:248] Engine 000: Avg prompt throughput: 415.3 tokens/s, Avg generation throughput: 537.6 tokens/s, Running: 32 reqs, Waiting: 247 reqs, GPU KV cache usage: 22.0%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:56:41 [loggers.py:248] Engine 000: Avg prompt throughput: 366.4 tokens/s, Avg generation throughput: 540.8 tokens/s, Running: 31 reqs, Waiting: 223 reqs, GPU KV cache usage: 18.9%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:56:51 [loggers.py:248] Engine 000: Avg prompt throughput: 537.1 tokens/s, Avg generation throughput: 537.6 tokens/s, Running: 32 reqs, Waiting: 202 reqs, GPU KV cache usage: 20.8%, Prefix cache hit rate: 0.0%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:57:01 [loggers.py:248] Engine 000: Avg prompt throughput: 686.9 tokens/s, Avg generation throughput: 505.6 tokens/s, Running: 32 reqs, Waiting: 167 reqs, GPU KV cache usage: 16.2%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:57:11 [loggers.py:248] Engine 000: Avg prompt throughput: 933.5 tokens/s, Avg generation throughput: 467.2 tokens/s, Running: 32 reqs, Waiting: 131 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:57:21 [loggers.py:248] Engine 000: Avg prompt throughput: 272.9 tokens/s, Avg generation throughput: 563.2 tokens/s, Running: 32 reqs, Waiting: 116 reqs, GPU KV cache usage: 19.7%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:57:31 [loggers.py:248] Engine 000: Avg prompt throughput: 591.2 tokens/s, Avg generation throughput: 515.2 tokens/s, Running: 32 reqs, Waiting: 91 reqs, GPU KV cache usage: 20.5%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:57:41 [loggers.py:248] Engine 000: Avg prompt throughput: 322.3 tokens/s, Avg generation throughput: 550.4 tokens/s, Running: 32 reqs, Waiting: 72 reqs, GPU KV cache usage: 19.0%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:57:51 [loggers.py:248] Engine 000: Avg prompt throughput: 481.7 tokens/s, Avg generation throughput: 531.2 tokens/s, Running: 31 reqs, Waiting: 49 reqs, GPU KV cache usage: 20.0%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:58:01 [loggers.py:248] Engine 000: Avg prompt throughput: 911.3 tokens/s, Avg generation throughput: 473.6 tokens/s, Running: 32 reqs, Waiting: 17 reqs, GPU KV cache usage: 19.9%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:58:11 [loggers.py:248] Engine 000: Avg prompt throughput: 264.0 tokens/s, Avg generation throughput: 544.2 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.4%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:58:21 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 347.9 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:58:31 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 130.4 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:58:41 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 43.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.1%, MM cache hit rate: 0.0%
+[0;36m(APIServer pid=12756)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=12756)[0;0m INFO 12-10 16:58:42 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP0 pid=13002)[0;0m INFO 12-10 16:58:42 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(Worker_TP1 pid=13003)[0;0m INFO 12-10 16:58:42 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(APIServer pid=12756)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=12756)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=12756)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json
new file mode 100644
index 0000000..d4a046d
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/RedHatAI_gemma-3-27b-it-FP8-dynamic_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 767.3010607799997,
+ "num_requests": 1000,
+ "total_num_tokens": 755432,
+ "requests_per_second": 1.303269408989804,
+ "tokens_per_second": 984.5314161719857
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..6b427c4
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-1e0af39c-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 198.24 \nTotal input tokens: 38358 \nTotal generated tokens: 40157 \nRequest throughput (req/s): 0.91 \nOutput token throughput (tok/s): 202.57 \nPeak output token throughput (tok/s): 311.00 \nPeak concurrent requests: 18.00 \nTotal Token throughput (tok/s): 396.07 \n---------------Time to First Token----------------\nMean TTFT (ms): 120.83 \nMedian TTFT (ms): 107.92 \nP99 TTFT (ms): 215.44 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 45.17 \nMedian TPOT (ms): 44.69 \nP99 TPOT (ms): 56.10 \n---------------Inter-token Latency----------------\nMean ITL (ms): 44.94 \nMedian ITL (ms): 42.97 \nP99 ITL (ms): 110.51 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..eeae5b3
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-5a7df568-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 249.37 \nTotal input tokens: 146694 \nTotal generated tokens: 153158 \nRequest throughput (req/s): 2.89 \nOutput token throughput (tok/s): 614.17 \nPeak output token throughput (tok/s): 832.00 \nPeak concurrent requests: 168.00 \nTotal Token throughput (tok/s): 1202.42 \n---------------Time to First Token----------------\nMean TTFT (ms): 14118.91 \nMedian TTFT (ms): 18607.33 \nP99 TTFT (ms): 31435.29 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 88.10 \nMedian TPOT (ms): 90.68 \nP99 TPOT (ms): 104.28 \n---------------Inter-token Latency----------------\nMean ITL (ms): 87.93 \nMedian ITL (ms): 84.69 \nP99 ITL (ms): 190.09 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
new file mode 100644
index 0000000..251da92
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
@@ -0,0 +1,1054 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:55:48 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:55:48 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=16712)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:55:52 [model.py:629] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:55:52 [model.py:1755] Using max model len 24576
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:55:52 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:56 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:56 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:38381 backend=nccl
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:56 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:57 [gpu_model_runner.py:3544] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:57 [compressed_tensors_wNa16.py:114] Using ConchLinearKernel for CompressedTensorsWNA16
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:57 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:57 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:55:57 [compressed_tensors_moe.py:189] Using CompressedTensorsWNA16MoEMethod
+[0;36m(EngineCore_DP0 pid=16880)[0;0m WARNING 12-09 19:55:57 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:01, 1.94it/s]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:02<00:02, 1.37s/it]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:04<00:01, 1.59s/it]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:06<00:00, 1.78s/it]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:06<00:00, 1.60s/it]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:56:04 [default_loader.py:308] Loading weights took 6.43 seconds
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:56:05 [gpu_model_runner.py:3626] Model loading took 16.2266 GiB memory and 7.385149 seconds
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:56:10 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/d30966f362/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:56:10 [backends.py:676] Dynamo bytecode transform time: 5.01 s
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:56:16 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=16880)[0;0m WARNING 12-09 19:56:19 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:57:02 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 49.43 s
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:57:02 [monitor.py:34] torch.compile takes 54.43 s in total
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:57:05 [gpu_worker.py:364] Available KV cache memory: 11.81 GiB
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:57:05 [kv_cache_utils.py:1287] GPU KV cache size: 128,976 tokens
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:57:05 [kv_cache_utils.py:1292] Maximum concurrency for 24,576 tokens per request: 5.25x
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 18.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 20.79it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 21.40it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 22.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 22.79it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 23.23it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 22.38it/s]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 26.17it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 27.79it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 26.71it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 26.90it/s]
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:57:07 [gpu_model_runner.py:4548] Graph capturing finished in 2 secs, took 1.24 GiB
+[0;36m(EngineCore_DP0 pid=16880)[0;0m INFO 12-09 19:57:07 [core.py:256] init engine (profile, create kv cache, warmup model) took 62.29 seconds
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:09 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=16712)[0;0m WARNING 12-09 19:57:09 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:09 [serving_responses.py:197] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [serving_completion.py:73] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:10 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=16712)[0;0m INFO: Started server process [16712]
+[0;36m(APIServer pid=16712)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=16712)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48882 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:30 [loggers.py:248] Engine 000: Avg prompt throughput: 43.9 tokens/s, Avg generation throughput: 53.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:40 [loggers.py:248] Engine 000: Avg prompt throughput: 176.9 tokens/s, Avg generation throughput: 123.5 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:57:50 [loggers.py:248] Engine 000: Avg prompt throughput: 207.4 tokens/s, Avg generation throughput: 180.4 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:58:00 [loggers.py:248] Engine 000: Avg prompt throughput: 215.3 tokens/s, Avg generation throughput: 188.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:58:10 [loggers.py:248] Engine 000: Avg prompt throughput: 285.9 tokens/s, Avg generation throughput: 174.8 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:58:20 [loggers.py:248] Engine 000: Avg prompt throughput: 310.2 tokens/s, Avg generation throughput: 215.5 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:58:30 [loggers.py:248] Engine 000: Avg prompt throughput: 452.9 tokens/s, Avg generation throughput: 220.2 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:58:40 [loggers.py:248] Engine 000: Avg prompt throughput: 185.7 tokens/s, Avg generation throughput: 199.3 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:58:50 [loggers.py:248] Engine 000: Avg prompt throughput: 289.2 tokens/s, Avg generation throughput: 225.9 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:59:00 [loggers.py:248] Engine 000: Avg prompt throughput: 320.1 tokens/s, Avg generation throughput: 264.1 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:59:10 [loggers.py:248] Engine 000: Avg prompt throughput: 131.0 tokens/s, Avg generation throughput: 283.9 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:57206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:59:20 [loggers.py:248] Engine 000: Avg prompt throughput: 111.2 tokens/s, Avg generation throughput: 271.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:59:30 [loggers.py:248] Engine 000: Avg prompt throughput: 245.5 tokens/s, Avg generation throughput: 224.7 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:59:40 [loggers.py:248] Engine 000: Avg prompt throughput: 184.9 tokens/s, Avg generation throughput: 241.5 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 19:59:50 [loggers.py:248] Engine 000: Avg prompt throughput: 130.9 tokens/s, Avg generation throughput: 237.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:00:00 [loggers.py:248] Engine 000: Avg prompt throughput: 80.4 tokens/s, Avg generation throughput: 148.3 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:00:10 [loggers.py:248] Engine 000: Avg prompt throughput: 140.2 tokens/s, Avg generation throughput: 181.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:41424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:00:20 [loggers.py:248] Engine 000: Avg prompt throughput: 243.2 tokens/s, Avg generation throughput: 179.2 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:37062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:00:30 [loggers.py:248] Engine 000: Avg prompt throughput: 82.2 tokens/s, Avg generation throughput: 249.9 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:00:40 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 146.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:00:50 [loggers.py:248] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 26.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:01:00 [loggers.py:248] Engine 000: Avg prompt throughput: 727.9 tokens/s, Avg generation throughput: 220.9 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 15.3%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:01:10 [loggers.py:248] Engine 000: Avg prompt throughput: 1149.0 tokens/s, Avg generation throughput: 467.9 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.7%, Prefix cache hit rate: 31.7%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:01:20 [loggers.py:248] Engine 000: Avg prompt throughput: 851.5 tokens/s, Avg generation throughput: 612.7 tokens/s, Running: 52 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.7%, Prefix cache hit rate: 39.9%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:01:30 [loggers.py:248] Engine 000: Avg prompt throughput: 641.7 tokens/s, Avg generation throughput: 700.6 tokens/s, Running: 64 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.6%, Prefix cache hit rate: 44.9%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:01:40 [loggers.py:248] Engine 000: Avg prompt throughput: 589.3 tokens/s, Avg generation throughput: 739.3 tokens/s, Running: 64 reqs, Waiting: 12 reqs, GPU KV cache usage: 19.5%, Prefix cache hit rate: 47.2%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:01:50 [loggers.py:248] Engine 000: Avg prompt throughput: 640.6 tokens/s, Avg generation throughput: 684.7 tokens/s, Running: 64 reqs, Waiting: 23 reqs, GPU KV cache usage: 18.8%, Prefix cache hit rate: 43.6%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:02:00 [loggers.py:248] Engine 000: Avg prompt throughput: 806.4 tokens/s, Avg generation throughput: 665.4 tokens/s, Running: 64 reqs, Waiting: 25 reqs, GPU KV cache usage: 19.7%, Prefix cache hit rate: 39.8%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:02:10 [loggers.py:248] Engine 000: Avg prompt throughput: 585.7 tokens/s, Avg generation throughput: 678.4 tokens/s, Running: 63 reqs, Waiting: 41 reqs, GPU KV cache usage: 19.0%, Prefix cache hit rate: 37.5%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:02:20 [loggers.py:248] Engine 000: Avg prompt throughput: 403.5 tokens/s, Avg generation throughput: 710.4 tokens/s, Running: 63 reqs, Waiting: 59 reqs, GPU KV cache usage: 17.3%, Prefix cache hit rate: 36.0%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:02:30 [loggers.py:248] Engine 000: Avg prompt throughput: 391.7 tokens/s, Avg generation throughput: 742.4 tokens/s, Running: 64 reqs, Waiting: 73 reqs, GPU KV cache usage: 17.0%, Prefix cache hit rate: 34.6%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:02:40 [loggers.py:248] Engine 000: Avg prompt throughput: 642.3 tokens/s, Avg generation throughput: 691.1 tokens/s, Running: 64 reqs, Waiting: 78 reqs, GPU KV cache usage: 14.3%, Prefix cache hit rate: 32.7%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:02:50 [loggers.py:248] Engine 000: Avg prompt throughput: 672.0 tokens/s, Avg generation throughput: 697.6 tokens/s, Running: 64 reqs, Waiting: 72 reqs, GPU KV cache usage: 14.8%, Prefix cache hit rate: 30.8%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:03:00 [loggers.py:248] Engine 000: Avg prompt throughput: 915.7 tokens/s, Avg generation throughput: 640.0 tokens/s, Running: 61 reqs, Waiting: 77 reqs, GPU KV cache usage: 16.6%, Prefix cache hit rate: 28.6%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:03:10 [loggers.py:248] Engine 000: Avg prompt throughput: 689.9 tokens/s, Avg generation throughput: 691.1 tokens/s, Running: 63 reqs, Waiting: 84 reqs, GPU KV cache usage: 15.2%, Prefix cache hit rate: 27.2%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:03:20 [loggers.py:248] Engine 000: Avg prompt throughput: 771.4 tokens/s, Avg generation throughput: 678.3 tokens/s, Running: 64 reqs, Waiting: 81 reqs, GPU KV cache usage: 16.4%, Prefix cache hit rate: 25.7%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45080 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:03:30 [loggers.py:248] Engine 000: Avg prompt throughput: 562.2 tokens/s, Avg generation throughput: 697.6 tokens/s, Running: 64 reqs, Waiting: 79 reqs, GPU KV cache usage: 15.8%, Prefix cache hit rate: 24.9%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:03:40 [loggers.py:248] Engine 000: Avg prompt throughput: 1037.6 tokens/s, Avg generation throughput: 659.2 tokens/s, Running: 62 reqs, Waiting: 73 reqs, GPU KV cache usage: 15.4%, Prefix cache hit rate: 23.3%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:51174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:42728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:45094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:43838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:54196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:55816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:48428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:46762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:34616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:44988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:35042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:03:50 [loggers.py:248] Engine 000: Avg prompt throughput: 638.7 tokens/s, Avg generation throughput: 684.7 tokens/s, Running: 64 reqs, Waiting: 100 reqs, GPU KV cache usage: 17.4%, Prefix cache hit rate: 22.4%
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:60412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO: 127.0.0.1:39778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:04:00 [loggers.py:248] Engine 000: Avg prompt throughput: 312.9 tokens/s, Avg generation throughput: 710.4 tokens/s, Running: 62 reqs, Waiting: 83 reqs, GPU KV cache usage: 16.2%, Prefix cache hit rate: 21.9%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:04:10 [loggers.py:248] Engine 000: Avg prompt throughput: 819.3 tokens/s, Avg generation throughput: 684.8 tokens/s, Running: 62 reqs, Waiting: 51 reqs, GPU KV cache usage: 18.6%, Prefix cache hit rate: 20.9%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:04:20 [loggers.py:248] Engine 000: Avg prompt throughput: 680.3 tokens/s, Avg generation throughput: 697.6 tokens/s, Running: 64 reqs, Waiting: 12 reqs, GPU KV cache usage: 16.6%, Prefix cache hit rate: 20.2%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:04:30 [loggers.py:248] Engine 000: Avg prompt throughput: 139.3 tokens/s, Avg generation throughput: 704.4 tokens/s, Running: 47 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.2%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:04:40 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 471.3 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.0%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:04:50 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 259.9 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:05:00 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 128.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=16712)[0;0m INFO 12-09 20:05:01 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1209 20:05:01.182181928 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
+[0;36m(APIServer pid=16712)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=16712)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=16712)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
new file mode 100644
index 0000000..a9d0dd3
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 726.8079903329999,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 1.3758792051004176,
+ "tokens_per_second": 1019.986034633913
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..38e1fc8
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-df1894e5-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 191.60 \nTotal input tokens: 38358 \nTotal generated tokens: 39514 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 206.23 \nPeak output token throughput (tok/s): 361.00 \nPeak concurrent requests: 15.00 \nTotal Token throughput (tok/s): 406.42 \n---------------Time to First Token----------------\nMean TTFT (ms): 74.14 \nMedian TTFT (ms): 66.00 \nP99 TTFT (ms): 145.78 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 28.00 \nMedian TPOT (ms): 28.14 \nP99 TPOT (ms): 35.12 \n---------------Inter-token Latency----------------\nMean ITL (ms): 27.59 \nMedian ITL (ms): 27.66 \nP99 ITL (ms): 54.69 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..78c75b6
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-e434292b-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 203.64 \nTotal input tokens: 146694 \nTotal generated tokens: 153347 \nRequest throughput (req/s): 3.54 \nOutput token throughput (tok/s): 753.01 \nPeak output token throughput (tok/s): 1160.00 \nPeak concurrent requests: 65.00 \nTotal Token throughput (tok/s): 1473.36 \n---------------Time to First Token----------------\nMean TTFT (ms): 93.13 \nMedian TTFT (ms): 87.23 \nP99 TTFT (ms): 181.31 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 47.81 \nMedian TPOT (ms): 47.62 \nP99 TPOT (ms): 61.13 \n---------------Inter-token Latency----------------\nMean ITL (ms): 47.43 \nMedian ITL (ms): 44.70 \nP99 ITL (ms): 109.90 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log
new file mode 100644
index 0000000..d5eea1f
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_server.log
@@ -0,0 +1,1076 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:50:38 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:50:38 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=43391)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:50:42 [model.py:629] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:50:42 [model.py:1755] Using max model len 24576
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:50:42 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=43554)[0;0m INFO 12-09 20:50:46 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=43554)[0;0m WARNING 12-09 20:50:46 [multiproc_executor.py:880] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+INFO 12-09 20:50:49 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:43809 backend=nccl
+INFO 12-09 20:50:49 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:43809 backend=nccl
+INFO 12-09 20:50:49 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-09 20:50:50 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-09 20:50:50 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:50:50 [gpu_model_runner.py:3544] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(Worker_TP1 pid=43637)[0;0m INFO 12-09 20:50:50 [compressed_tensors_wNa16.py:114] Using ConchLinearKernel for CompressedTensorsWNA16
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:50:50 [compressed_tensors_wNa16.py:114] Using ConchLinearKernel for CompressedTensorsWNA16
+[0;36m(Worker_TP1 pid=43637)[0;0m INFO 12-09 20:50:50 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=43637)[0;0m INFO 12-09 20:50:50 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP1 pid=43637)[0;0m INFO 12-09 20:50:50 [compressed_tensors_moe.py:189] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP1 pid=43637)[0;0m WARNING 12-09 20:50:50 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:50:50 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:50:50 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:50:50 [compressed_tensors_moe.py:189] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP0 pid=43636)[0;0m WARNING 12-09 20:50:50 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=43636)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=43636)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:01, 2.20it/s]
+[0;36m(Worker_TP0 pid=43636)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:02<00:02, 1.34s/it]
+[0;36m(Worker_TP0 pid=43636)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:04<00:01, 1.55s/it]
+[0;36m(Worker_TP0 pid=43636)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:06<00:00, 1.75s/it]
+[0;36m(Worker_TP0 pid=43636)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:06<00:00, 1.57s/it]
+[0;36m(Worker_TP0 pid=43636)[0;0m
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:50:57 [default_loader.py:308] Loading weights took 6.31 seconds
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:50:58 [gpu_model_runner.py:3626] Model loading took 8.1992 GiB memory and 7.083759 seconds
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:51:04 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/b5f601a270/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:51:04 [backends.py:676] Dynamo bytecode transform time: 5.57 s
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:51:07 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP1 pid=43637)[0;0m INFO 12-09 20:51:07 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP1 pid=43637)[0;0m WARNING 12-09 20:51:07 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP0 pid=43636)[0;0m WARNING 12-09 20:51:07 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:51:09 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 2.69 s
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:51:09 [monitor.py:34] torch.compile takes 8.26 s in total
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:51:12 [gpu_worker.py:364] Available KV cache memory: 19.36 GiB
+[0;36m(EngineCore_DP0 pid=43554)[0;0m INFO 12-09 20:51:12 [kv_cache_utils.py:1287] GPU KV cache size: 422,960 tokens
+[0;36m(EngineCore_DP0 pid=43554)[0;0m INFO 12-09 20:51:12 [kv_cache_utils.py:1292] Maximum concurrency for 24,576 tokens per request: 17.21x
+[0;36m(Worker_TP0 pid=43636)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:07, 2.41it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:06, 2.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:01<00:06, 2.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:01<00:06, 2.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:02<00:05, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:02<00:05, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:02<00:04, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:03<00:04, 2.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:03<00:04, 2.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:04<00:03, 2.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:04<00:03, 2.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:04<00:02, 2.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:05<00:02, 2.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:05<00:02, 2.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:06<00:01, 2.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:06<00:01, 2.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:06<00:00, 2.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:07<00:00, 2.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.42it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:07<00:00, 2.44it/s]
+[0;36m(Worker_TP0 pid=43636)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:04, 2.48it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:03, 2.49it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:01<00:03, 2.50it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:01<00:02, 2.50it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:01<00:02, 2.51it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:02<00:01, 2.52it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:02<00:01, 2.52it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:03<00:01, 2.51it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:03<00:00, 2.51it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:03<00:00, 2.52it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.50it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.51it/s]
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:51:25 [gpu_model_runner.py:4548] Graph capturing finished in 13 secs, took 1.58 GiB
+[0;36m(EngineCore_DP0 pid=43554)[0;0m INFO 12-09 20:51:25 [core.py:256] init engine (profile, create kv cache, warmup model) took 27.07 seconds
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=43391)[0;0m WARNING 12-09 20:51:27 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [serving_responses.py:197] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [serving_completion.py:73] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:27 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=43391)[0;0m INFO: Started server process [43391]
+[0;36m(APIServer pid=43391)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=43391)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55700 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:47 [loggers.py:248] Engine 000: Avg prompt throughput: 84.0 tokens/s, Avg generation throughput: 84.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:51:57 [loggers.py:248] Engine 000: Avg prompt throughput: 136.8 tokens/s, Avg generation throughput: 160.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:52:07 [loggers.py:248] Engine 000: Avg prompt throughput: 207.4 tokens/s, Avg generation throughput: 203.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:52:17 [loggers.py:248] Engine 000: Avg prompt throughput: 224.2 tokens/s, Avg generation throughput: 157.4 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:52:27 [loggers.py:248] Engine 000: Avg prompt throughput: 277.0 tokens/s, Avg generation throughput: 187.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:52:37 [loggers.py:248] Engine 000: Avg prompt throughput: 310.2 tokens/s, Avg generation throughput: 220.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:51904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:52:47 [loggers.py:248] Engine 000: Avg prompt throughput: 452.9 tokens/s, Avg generation throughput: 209.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:34918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:52:57 [loggers.py:248] Engine 000: Avg prompt throughput: 185.7 tokens/s, Avg generation throughput: 190.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:53:07 [loggers.py:248] Engine 000: Avg prompt throughput: 299.2 tokens/s, Avg generation throughput: 231.9 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:53:17 [loggers.py:248] Engine 000: Avg prompt throughput: 310.1 tokens/s, Avg generation throughput: 334.7 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:49820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:53:27 [loggers.py:248] Engine 000: Avg prompt throughput: 132.9 tokens/s, Avg generation throughput: 250.4 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:53:37 [loggers.py:248] Engine 000: Avg prompt throughput: 109.9 tokens/s, Avg generation throughput: 201.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:53:47 [loggers.py:248] Engine 000: Avg prompt throughput: 246.6 tokens/s, Avg generation throughput: 207.5 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:41472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:43318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:53:57 [loggers.py:248] Engine 000: Avg prompt throughput: 184.0 tokens/s, Avg generation throughput: 258.4 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:40546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:54:07 [loggers.py:248] Engine 000: Avg prompt throughput: 130.1 tokens/s, Avg generation throughput: 253.5 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:41472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:54:17 [loggers.py:248] Engine 000: Avg prompt throughput: 80.4 tokens/s, Avg generation throughput: 80.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:54:27 [loggers.py:248] Engine 000: Avg prompt throughput: 140.2 tokens/s, Avg generation throughput: 222.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:58252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:58260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:58260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:58274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:58286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:58286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:55384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:54:37 [loggers.py:248] Engine 000: Avg prompt throughput: 275.0 tokens/s, Avg generation throughput: 212.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:59324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:54:47 [loggers.py:248] Engine 000: Avg prompt throughput: 50.4 tokens/s, Avg generation throughput: 220.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:54:57 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 74.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:55:07 [loggers.py:248] Engine 000: Avg prompt throughput: 434.0 tokens/s, Avg generation throughput: 201.0 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 9.7%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:55:17 [loggers.py:248] Engine 000: Avg prompt throughput: 1294.3 tokens/s, Avg generation throughput: 581.5 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 29.9%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:55:27 [loggers.py:248] Engine 000: Avg prompt throughput: 892.1 tokens/s, Avg generation throughput: 758.8 tokens/s, Running: 34 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 39.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:55:37 [loggers.py:248] Engine 000: Avg prompt throughput: 670.6 tokens/s, Avg generation throughput: 868.0 tokens/s, Running: 44 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 44.4%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:55:47 [loggers.py:248] Engine 000: Avg prompt throughput: 671.7 tokens/s, Avg generation throughput: 842.1 tokens/s, Running: 38 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 47.2%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:55:57 [loggers.py:248] Engine 000: Avg prompt throughput: 943.0 tokens/s, Avg generation throughput: 802.8 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 42.1%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:56:07 [loggers.py:248] Engine 000: Avg prompt throughput: 1003.2 tokens/s, Avg generation throughput: 897.9 tokens/s, Running: 48 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 37.8%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:56:17 [loggers.py:248] Engine 000: Avg prompt throughput: 434.6 tokens/s, Avg generation throughput: 840.7 tokens/s, Running: 41 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 36.2%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:56:27 [loggers.py:248] Engine 000: Avg prompt throughput: 968.1 tokens/s, Avg generation throughput: 976.4 tokens/s, Running: 61 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 33.0%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:56:37 [loggers.py:248] Engine 000: Avg prompt throughput: 629.3 tokens/s, Avg generation throughput: 1008.7 tokens/s, Running: 47 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 31.3%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:56:47 [loggers.py:248] Engine 000: Avg prompt throughput: 1077.6 tokens/s, Avg generation throughput: 833.7 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 29.2%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:56:57 [loggers.py:248] Engine 000: Avg prompt throughput: 635.2 tokens/s, Avg generation throughput: 744.8 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 27.8%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:57:07 [loggers.py:248] Engine 000: Avg prompt throughput: 766.8 tokens/s, Avg generation throughput: 721.7 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 26.3%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:57:17 [loggers.py:248] Engine 000: Avg prompt throughput: 661.1 tokens/s, Avg generation throughput: 823.3 tokens/s, Running: 38 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 25.3%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:56532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:57:27 [loggers.py:248] Engine 000: Avg prompt throughput: 903.0 tokens/s, Avg generation throughput: 695.2 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 23.9%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:35308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:57:37 [loggers.py:248] Engine 000: Avg prompt throughput: 759.7 tokens/s, Avg generation throughput: 740.9 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 22.8%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:57:47 [loggers.py:248] Engine 000: Avg prompt throughput: 533.6 tokens/s, Avg generation throughput: 815.8 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 22.1%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47734 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:47730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:57:57 [loggers.py:248] Engine 000: Avg prompt throughput: 1133.1 tokens/s, Avg generation throughput: 780.8 tokens/s, Running: 41 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:45022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:38392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:42392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:44982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO: 127.0.0.1:53246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:58:07 [loggers.py:248] Engine 000: Avg prompt throughput: 259.0 tokens/s, Avg generation throughput: 882.2 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 20.4%
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:58:17 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 411.8 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 20.4%
+[0;36m(APIServer pid=43391)[0;0m INFO 12-09 20:58:24 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP0 pid=43636)[0;0m INFO 12-09 20:58:24 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(Worker_TP1 pid=43637)[0;0m INFO 12-09 20:58:24 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(APIServer pid=43391)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=43391)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=43391)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json
new file mode 100644
index 0000000..d5a7613
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 449.04138348200104,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 2.2269662369327774,
+ "tokens_per_second": 1650.9257882903235
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..912892b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-b53e2ad8-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 196.26 \nTotal input tokens: 38358 \nTotal generated tokens: 38404 \nRequest throughput (req/s): 0.92 \nOutput token throughput (tok/s): 195.68 \nPeak output token throughput (tok/s): 304.00 \nPeak concurrent requests: 18.00 \nTotal Token throughput (tok/s): 391.12 \n---------------Time to First Token----------------\nMean TTFT (ms): 519.96 \nMedian TTFT (ms): 124.31 \nP99 TTFT (ms): 10120.96 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 44.04 \nMedian TPOT (ms): 42.30 \nP99 TPOT (ms): 57.88 \n---------------Inter-token Latency----------------\nMean ITL (ms): 43.67 \nMedian ITL (ms): 40.54 \nP99 ITL (ms): 125.33 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..1dc37a9
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-72f2dd3c-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 351.46 \nTotal input tokens: 146694 \nTotal generated tokens: 148389 \nRequest throughput (req/s): 2.05 \nOutput token throughput (tok/s): 422.20 \nPeak output token throughput (tok/s): 544.00 \nPeak concurrent requests: 369.00 \nTotal Token throughput (tok/s): 839.58 \n---------------Time to First Token----------------\nMean TTFT (ms): 72209.01 \nMedian TTFT (ms): 79268.59 \nP99 TTFT (ms): 146128.91 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 71.26 \nMedian TPOT (ms): 71.50 \nP99 TPOT (ms): 93.20 \n---------------Inter-token Latency----------------\nMean ITL (ms): 70.85 \nMedian ITL (ms): 62.03 \nP99 ITL (ms): 206.26 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log
new file mode 100644
index 0000000..c17729a
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_server.log
@@ -0,0 +1,1128 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:35 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:35 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', 'trust_remote_code': True, 'max_model_len': 16384, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 32}
+[0;36m(APIServer pid=22068)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:38 [model.py:629] Resolved architecture: Qwen3NextForCausalLM
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:38 [model.py:1755] Using max model len 16384
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:38 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:38 [config.py:310] Disabling cascade attention since it is not supported for hybrid models.
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:38 [config.py:437] Setting attention block size to 544 tokens to ensure that attention page size is >= mamba page size.
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:20:38 [config.py:461] Padding mamba page size by 1.49% to ensure that mamba page size and attention page size are exactly equal.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=22230)[0;0m INFO 12-11 20:20:42 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=16384, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit, enable_prefix_caching=False, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 64, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=22230)[0;0m WARNING 12-11 20:20:42 [multiproc_executor.py:880] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+INFO 12-11 20:20:45 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:53731 backend=nccl
+INFO 12-11 20:20:45 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:53731 backend=nccl
+INFO 12-11 20:20:45 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-11 20:20:46 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-11 20:20:46 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:20:46 [gpu_model_runner.py:3544] Starting to load model cpatonn/Qwen3-Next-80B-A3B-Instruct-AWQ-4bit...
+[0;36m(Worker_TP1 pid=22313)[0;0m WARNING 12-11 20:20:47 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP1 pid=22313)[0;0m INFO 12-11 20:20:47 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP1 pid=22313)[0;0m INFO 12-11 20:20:47 [compressed_tensors_moe.py:189] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP0 pid=22312)[0;0m WARNING 12-11 20:20:47 [compressed_tensors.py:742] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:20:47 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:20:47 [compressed_tensors_moe.py:189] Using CompressedTensorsWNA16MoEMethod
+[0;36m(Worker_TP1 pid=22313)[0;0m INFO 12-11 20:20:47 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:20:47 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/10 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 10% Completed | 1/10 [00:03<00:29, 3.32s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 20% Completed | 2/10 [00:06<00:26, 3.30s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 30% Completed | 3/10 [00:10<00:23, 3.43s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 40% Completed | 4/10 [00:13<00:21, 3.52s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 5/10 [00:16<00:16, 3.33s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 60% Completed | 6/10 [00:20<00:13, 3.36s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 70% Completed | 7/10 [00:23<00:10, 3.37s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 80% Completed | 8/10 [00:27<00:06, 3.38s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 10/10 [00:30<00:00, 2.64s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 10/10 [00:30<00:00, 3.07s/it]
+[0;36m(Worker_TP0 pid=22312)[0;0m
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:18 [default_loader.py:308] Loading weights took 30.72 seconds
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:19 [gpu_model_runner.py:3626] Model loading took 23.5020 GiB memory and 31.787280 seconds
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:23 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/b8cc2bf5b2/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:23 [backends.py:676] Dynamo bytecode transform time: 4.21 s
+[0;36m(Worker_TP1 pid=22313)[0;0m INFO 12-11 20:21:25 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:25 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=22312)[0;0m WARNING 12-11 20:21:26 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP1 pid=22313)[0;0m WARNING 12-11 20:21:26 [fused_moe.py:888] Using default MoE config. Performance might be sub-optimal! Config file not found at ['/opt/venv/lib/python3.13/site-packages/vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=AMD-gfx1201,dtype=int4_w4a16.json']
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:31 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 5.37 s
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:31 [monitor.py:34] torch.compile takes 9.58 s in total
+[0;36m(Worker_TP0 pid=22312)[0;0m WARNING 12-11 20:21:31 [decorators.py:509] Cannot save aot compilation to path /home/kyuz0/.cache/vllm/torch_aot_compile/b81e0a6e575f55d4e244cb6958a5419624d6ffb67d44abd043f8d86e42bb593b/rank_0_0/model, error:
+[0;36m(Worker_TP1 pid=22313)[0;0m WARNING 12-11 20:21:31 [decorators.py:509] Cannot save aot compilation to path /home/kyuz0/.cache/vllm/torch_aot_compile/b81e0a6e575f55d4e244cb6958a5419624d6ffb67d44abd043f8d86e42bb593b/rank_1_0/model, error:
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:33 [gpu_worker.py:364] Available KV cache memory: 7.09 GiB
+[0;36m(EngineCore_DP0 pid=22230)[0;0m INFO 12-11 20:21:34 [kv_cache_utils.py:1287] GPU KV cache size: 154,496 tokens
+[0;36m(EngineCore_DP0 pid=22230)[0;0m INFO 12-11 20:21:34 [kv_cache_utils.py:1292] Maximum concurrency for 16,384 tokens per request: 33.47x
+[0;36m(Worker_TP0 pid=22312)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 9%|▉ | 1/11 [00:00<00:05, 1.78it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 2/11 [00:01<00:04, 1.80it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 3/11 [00:01<00:04, 1.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▋ | 4/11 [00:02<00:03, 1.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▌ | 5/11 [00:02<00:03, 1.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▍ | 6/11 [00:03<00:02, 1.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▎ | 7/11 [00:03<00:01, 2.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 8/11 [00:04<00:01, 2.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 9/11 [00:04<00:00, 2.05it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 91%|█████████ | 10/11 [00:05<00:00, 2.07it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 2.11it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:05<00:00, 2.00it/s]
+[0;36m(Worker_TP0 pid=22312)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 14%|█▍ | 1/7 [00:00<00:03, 1.85it/s]
Capturing CUDA graphs (decode, FULL): 29%|██▊ | 2/7 [00:01<00:02, 1.90it/s]
Capturing CUDA graphs (decode, FULL): 43%|████▎ | 3/7 [00:01<00:01, 2.04it/s]
Capturing CUDA graphs (decode, FULL): 57%|█████▋ | 4/7 [00:01<00:01, 2.13it/s]
Capturing CUDA graphs (decode, FULL): 71%|███████▏ | 5/7 [00:02<00:00, 2.17it/s]
Capturing CUDA graphs (decode, FULL): 86%|████████▌ | 6/7 [00:02<00:00, 2.19it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:03<00:00, 2.14it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:03<00:00, 2.10it/s]
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:21:43 [gpu_model_runner.py:4548] Graph capturing finished in 10 secs, took 0.87 GiB
+[0;36m(EngineCore_DP0 pid=22230)[0;0m INFO 12-11 20:21:43 [core.py:256] init engine (profile, create kv cache, warmup model) took 24.45 seconds
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=22068)[0;0m WARNING 12-11 20:21:45 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [serving_responses.py:197] Using default chat sampling params from model: {'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:21:45 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=22068)[0;0m INFO: Started server process [22068]
+[0;36m(APIServer pid=22068)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=22068)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52410 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP0 pid=22312)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (12) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP1 pid=22313)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (12) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP0 pid=22312)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(Worker_TP1 pid=22313)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:22:16 [loggers.py:248] Engine 000: Avg prompt throughput: 2.4 tokens/s, Avg generation throughput: 18.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:22:26 [loggers.py:248] Engine 000: Avg prompt throughput: 218.4 tokens/s, Avg generation throughput: 17.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:22:36 [loggers.py:248] Engine 000: Avg prompt throughput: 207.4 tokens/s, Avg generation throughput: 244.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:22:46 [loggers.py:248] Engine 000: Avg prompt throughput: 215.3 tokens/s, Avg generation throughput: 201.8 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:22:56 [loggers.py:248] Engine 000: Avg prompt throughput: 216.4 tokens/s, Avg generation throughput: 191.4 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP0 pid=22312)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (14) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP1 pid=22313)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (14) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP0 pid=22312)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(Worker_TP1 pid=22313)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP0 pid=22312)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (11) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP0 pid=22312)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(Worker_TP1 pid=22313)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (11) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP1 pid=22313)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP0 pid=22312)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (13) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP0 pid=22312)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(Worker_TP1 pid=22313)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (13) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP1 pid=22313)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP1 pid=22313)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (15) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP0 pid=22312)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (15) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP1 pid=22313)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(Worker_TP0 pid=22312)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:23:06 [loggers.py:248] Engine 000: Avg prompt throughput: 379.7 tokens/s, Avg generation throughput: 174.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:23:16 [loggers.py:248] Engine 000: Avg prompt throughput: 452.9 tokens/s, Avg generation throughput: 186.8 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:23:26 [loggers.py:248] Engine 000: Avg prompt throughput: 174.4 tokens/s, Avg generation throughput: 197.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:23:36 [loggers.py:248] Engine 000: Avg prompt throughput: 287.3 tokens/s, Avg generation throughput: 211.8 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:23:46 [loggers.py:248] Engine 000: Avg prompt throughput: 333.3 tokens/s, Avg generation throughput: 253.0 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:23:56 [loggers.py:248] Engine 000: Avg prompt throughput: 131.0 tokens/s, Avg generation throughput: 273.9 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:24:06 [loggers.py:248] Engine 000: Avg prompt throughput: 111.2 tokens/s, Avg generation throughput: 274.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:24:16 [loggers.py:248] Engine 000: Avg prompt throughput: 243.1 tokens/s, Avg generation throughput: 225.3 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:33806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:24:26 [loggers.py:248] Engine 000: Avg prompt throughput: 131.6 tokens/s, Avg generation throughput: 242.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:24:36 [loggers.py:248] Engine 000: Avg prompt throughput: 186.6 tokens/s, Avg generation throughput: 219.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:24:46 [loggers.py:248] Engine 000: Avg prompt throughput: 46.1 tokens/s, Avg generation throughput: 149.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP1 pid=22313)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (10) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP0 pid=22312)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (10) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP1 pid=22313)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(Worker_TP0 pid=22312)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:24:56 [loggers.py:248] Engine 000: Avg prompt throughput: 170.7 tokens/s, Avg generation throughput: 185.5 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:37250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:25:06 [loggers.py:248] Engine 000: Avg prompt throughput: 240.5 tokens/s, Avg generation throughput: 181.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:52834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:25:16 [loggers.py:248] Engine 000: Avg prompt throughput: 88.7 tokens/s, Avg generation throughput: 250.8 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:25:26 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 143.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(Worker_TP0 pid=22312)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (12) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP0 pid=22312)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(Worker_TP1 pid=22313)[0;0m /opt/venv/lib64/python3.13/site-packages/vllm/model_executor/layers/fla/ops/utils.py:113: UserWarning: Input tensor shape suggests potential format mismatch: seq_len (12) < num_heads (16). This may indicate the inputs were passed in head-first format [B, H, T, ...] when head_first=False was specified. Please verify your input tensor format matches the expected shape [B, T, H, ...].
+[0;36m(Worker_TP1 pid=22313)[0;0m return fn(*contiguous_args, **contiguous_kwargs)
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:25:36 [loggers.py:248] Engine 000: Avg prompt throughput: 4.8 tokens/s, Avg generation throughput: 23.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:25:46 [loggers.py:248] Engine 000: Avg prompt throughput: 853.4 tokens/s, Avg generation throughput: 216.3 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:25:56 [loggers.py:248] Engine 000: Avg prompt throughput: 830.1 tokens/s, Avg generation throughput: 399.2 tokens/s, Running: 32 reqs, Waiting: 20 reqs, GPU KV cache usage: 11.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:26:06 [loggers.py:248] Engine 000: Avg prompt throughput: 493.8 tokens/s, Avg generation throughput: 422.4 tokens/s, Running: 31 reqs, Waiting: 30 reqs, GPU KV cache usage: 11.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:26:16 [loggers.py:248] Engine 000: Avg prompt throughput: 438.2 tokens/s, Avg generation throughput: 441.6 tokens/s, Running: 32 reqs, Waiting: 48 reqs, GPU KV cache usage: 12.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:26:26 [loggers.py:248] Engine 000: Avg prompt throughput: 261.3 tokens/s, Avg generation throughput: 480.0 tokens/s, Running: 32 reqs, Waiting: 84 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:26:36 [loggers.py:248] Engine 000: Avg prompt throughput: 315.7 tokens/s, Avg generation throughput: 454.4 tokens/s, Running: 31 reqs, Waiting: 106 reqs, GPU KV cache usage: 11.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:26:46 [loggers.py:248] Engine 000: Avg prompt throughput: 499.9 tokens/s, Avg generation throughput: 425.5 tokens/s, Running: 32 reqs, Waiting: 107 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:26:56 [loggers.py:248] Engine 000: Avg prompt throughput: 278.9 tokens/s, Avg generation throughput: 464.0 tokens/s, Running: 32 reqs, Waiting: 137 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:58032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:27:06 [loggers.py:248] Engine 000: Avg prompt throughput: 410.1 tokens/s, Avg generation throughput: 444.8 tokens/s, Running: 32 reqs, Waiting: 163 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49866 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49968 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42046 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:27:16 [loggers.py:248] Engine 000: Avg prompt throughput: 461.3 tokens/s, Avg generation throughput: 441.6 tokens/s, Running: 32 reqs, Waiting: 183 reqs, GPU KV cache usage: 12.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:53200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:27:26 [loggers.py:248] Engine 000: Avg prompt throughput: 463.6 tokens/s, Avg generation throughput: 422.4 tokens/s, Running: 31 reqs, Waiting: 200 reqs, GPU KV cache usage: 11.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47316 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:27:36 [loggers.py:248] Engine 000: Avg prompt throughput: 682.2 tokens/s, Avg generation throughput: 409.6 tokens/s, Running: 32 reqs, Waiting: 200 reqs, GPU KV cache usage: 11.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42274 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54030 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:27:46 [loggers.py:248] Engine 000: Avg prompt throughput: 123.6 tokens/s, Avg generation throughput: 489.6 tokens/s, Running: 31 reqs, Waiting: 231 reqs, GPU KV cache usage: 11.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:46516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:27:56 [loggers.py:248] Engine 000: Avg prompt throughput: 349.4 tokens/s, Avg generation throughput: 457.6 tokens/s, Running: 32 reqs, Waiting: 254 reqs, GPU KV cache usage: 11.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42596 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42652 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:42670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:28:06 [loggers.py:248] Engine 000: Avg prompt throughput: 307.0 tokens/s, Avg generation throughput: 454.4 tokens/s, Running: 32 reqs, Waiting: 270 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47348 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:28:16 [loggers.py:248] Engine 000: Avg prompt throughput: 450.5 tokens/s, Avg generation throughput: 438.4 tokens/s, Running: 31 reqs, Waiting: 279 reqs, GPU KV cache usage: 11.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:36472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:40876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:50800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:28:26 [loggers.py:248] Engine 000: Avg prompt throughput: 206.5 tokens/s, Avg generation throughput: 470.4 tokens/s, Running: 32 reqs, Waiting: 299 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57778 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:39200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:55272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:57892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:60898 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:60914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:60924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:60932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:54088 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:44862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO: 127.0.0.1:49912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:28:36 [loggers.py:248] Engine 000: Avg prompt throughput: 470.7 tokens/s, Avg generation throughput: 448.0 tokens/s, Running: 32 reqs, Waiting: 331 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:28:46 [loggers.py:248] Engine 000: Avg prompt throughput: 496.4 tokens/s, Avg generation throughput: 441.6 tokens/s, Running: 32 reqs, Waiting: 306 reqs, GPU KV cache usage: 11.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:28:56 [loggers.py:248] Engine 000: Avg prompt throughput: 849.8 tokens/s, Avg generation throughput: 403.2 tokens/s, Running: 32 reqs, Waiting: 279 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:29:06 [loggers.py:248] Engine 000: Avg prompt throughput: 371.2 tokens/s, Avg generation throughput: 448.0 tokens/s, Running: 32 reqs, Waiting: 258 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:29:16 [loggers.py:248] Engine 000: Avg prompt throughput: 324.2 tokens/s, Avg generation throughput: 454.4 tokens/s, Running: 32 reqs, Waiting: 236 reqs, GPU KV cache usage: 11.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:29:26 [loggers.py:248] Engine 000: Avg prompt throughput: 506.1 tokens/s, Avg generation throughput: 438.4 tokens/s, Running: 32 reqs, Waiting: 215 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:29:36 [loggers.py:248] Engine 000: Avg prompt throughput: 117.4 tokens/s, Avg generation throughput: 464.0 tokens/s, Running: 32 reqs, Waiting: 199 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:29:46 [loggers.py:248] Engine 000: Avg prompt throughput: 476.2 tokens/s, Avg generation throughput: 428.8 tokens/s, Running: 32 reqs, Waiting: 175 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:29:56 [loggers.py:248] Engine 000: Avg prompt throughput: 851.7 tokens/s, Avg generation throughput: 387.2 tokens/s, Running: 32 reqs, Waiting: 144 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:30:06 [loggers.py:248] Engine 000: Avg prompt throughput: 287.1 tokens/s, Avg generation throughput: 441.6 tokens/s, Running: 32 reqs, Waiting: 127 reqs, GPU KV cache usage: 11.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:30:16 [loggers.py:248] Engine 000: Avg prompt throughput: 560.8 tokens/s, Avg generation throughput: 409.6 tokens/s, Running: 31 reqs, Waiting: 102 reqs, GPU KV cache usage: 11.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:30:26 [loggers.py:248] Engine 000: Avg prompt throughput: 264.6 tokens/s, Avg generation throughput: 457.6 tokens/s, Running: 32 reqs, Waiting: 90 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:30:36 [loggers.py:248] Engine 000: Avg prompt throughput: 286.2 tokens/s, Avg generation throughput: 460.8 tokens/s, Running: 32 reqs, Waiting: 70 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:30:46 [loggers.py:248] Engine 000: Avg prompt throughput: 866.7 tokens/s, Avg generation throughput: 400.0 tokens/s, Running: 32 reqs, Waiting: 39 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:30:56 [loggers.py:248] Engine 000: Avg prompt throughput: 340.5 tokens/s, Avg generation throughput: 448.0 tokens/s, Running: 31 reqs, Waiting: 16 reqs, GPU KV cache usage: 11.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:31:06 [loggers.py:248] Engine 000: Avg prompt throughput: 170.5 tokens/s, Avg generation throughput: 475.3 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:31:16 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 359.4 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:31:26 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 133.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=22068)[0;0m INFO 12-11 20:31:27 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP0 pid=22312)[0;0m INFO 12-11 20:31:27 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(Worker_TP1 pid=22313)[0;0m INFO 12-11 20:31:27 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(APIServer pid=22068)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=22068)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=22068)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_throughput.json
new file mode 100644
index 0000000..7e1a289
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/cpatonn_Qwen3-Next-80B-A3B-Instruct-AWQ-4bit_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 1089.8343978980001,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 0.9175705978162676,
+ "tokens_per_second": 680.226281561525
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..ed45de5
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-df39931f-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 198.49 \nTotal input tokens: 37841 \nTotal generated tokens: 38900 \nRequest throughput (req/s): 0.91 \nOutput token throughput (tok/s): 195.98 \nPeak output token throughput (tok/s): 365.00 \nPeak concurrent requests: 16.00 \nTotal Token throughput (tok/s): 386.62 \n---------------Time to First Token----------------\nMean TTFT (ms): 80.34 \nMedian TTFT (ms): 69.07 \nP99 TTFT (ms): 180.64 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 37.10 \nMedian TPOT (ms): 37.08 \nP99 TPOT (ms): 40.10 \n---------------Inter-token Latency----------------\nMean ITL (ms): 37.10 \nMedian ITL (ms): 36.27 \nP99 ITL (ms): 58.47 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..8594ee8
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-ba51e591-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 719 \nFailed requests: 1 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 209.75 \nTotal input tokens: 145579 \nTotal generated tokens: 150640 \nRequest throughput (req/s): 3.43 \nOutput token throughput (tok/s): 718.18 \nPeak output token throughput (tok/s): 1259.00 \nPeak concurrent requests: 58.00 \nTotal Token throughput (tok/s): 1412.23 \n---------------Time to First Token----------------\nMean TTFT (ms): 85.92 \nMedian TTFT (ms): 75.54 \nP99 TTFT (ms): 188.35 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 42.81 \nMedian TPOT (ms): 42.63 \nP99 TPOT (ms): 54.28 \n---------------Inter-token Latency----------------\nMean ITL (ms): 42.63 \nMedian ITL (ms): 40.26 \nP99 ITL (ms): 125.13 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
new file mode 100644
index 0000000..87e7888
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
@@ -0,0 +1,1043 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:09 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:09 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 65536, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:13 [model.py:629] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:13 [model.py:1755] Using max model len 65536
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:13 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:17 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=65536, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:17 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:51127 backend=nccl
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:17 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:18 [gpu_model_runner.py:3544] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:18 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 5.39it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.58it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.25it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.14it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.28it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:22 [default_loader.py:308] Loading weights took 3.13 seconds
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:23 [gpu_model_runner.py:3626] Model loading took 15.0586 GiB memory and 4.320186 seconds
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:25 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/dfa108d3b7/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:25 [backends.py:676] Dynamo bytecode transform time: 2.05 s
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:29 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:39 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 12.70 s
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:39 [monitor.py:34] torch.compile takes 14.75 s in total
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:41 [gpu_worker.py:364] Available KV cache memory: 12.73 GiB
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:41 [kv_cache_utils.py:1287] GPU KV cache size: 104,320 tokens
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:41 [kv_cache_utils.py:1292] Maximum concurrency for 65,536 tokens per request: 1.59x
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 19.11it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 20.72it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 21.20it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 21.61it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 21.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 22.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 21.91it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 21.69it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 23.00it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 22.61it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 22.60it/s]
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:43 [gpu_model_runner.py:4548] Graph capturing finished in 2 secs, took 1.36 GiB
+[0;36m(EngineCore_DP0 pid=4113)[0;0m INFO 12-09 18:49:43 [core.py:256] init engine (profile, create kv cache, warmup model) took 20.45 seconds
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:45 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=3950)[0;0m WARNING 12-09 18:49:45 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:45 [serving_responses.py:197] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:45 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:45 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:49:46 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=3950)[0;0m INFO: Started server process [3950]
+[0;36m(APIServer pid=3950)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=3950)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42160 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:50:06 [loggers.py:248] Engine 000: Avg prompt throughput: 41.7 tokens/s, Avg generation throughput: 36.0 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:50:16 [loggers.py:248] Engine 000: Avg prompt throughput: 135.3 tokens/s, Avg generation throughput: 115.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:50:26 [loggers.py:248] Engine 000: Avg prompt throughput: 183.3 tokens/s, Avg generation throughput: 183.7 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:50:36 [loggers.py:248] Engine 000: Avg prompt throughput: 278.0 tokens/s, Avg generation throughput: 182.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:50:46 [loggers.py:248] Engine 000: Avg prompt throughput: 191.1 tokens/s, Avg generation throughput: 177.5 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:50:56 [loggers.py:248] Engine 000: Avg prompt throughput: 365.3 tokens/s, Avg generation throughput: 206.6 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51322 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:51:06 [loggers.py:248] Engine 000: Avg prompt throughput: 347.5 tokens/s, Avg generation throughput: 230.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:51:16 [loggers.py:248] Engine 000: Avg prompt throughput: 167.3 tokens/s, Avg generation throughput: 197.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:51:26 [loggers.py:248] Engine 000: Avg prompt throughput: 376.3 tokens/s, Avg generation throughput: 226.3 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:51:36 [loggers.py:248] Engine 000: Avg prompt throughput: 223.3 tokens/s, Avg generation throughput: 285.7 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:51:46 [loggers.py:248] Engine 000: Avg prompt throughput: 276.0 tokens/s, Avg generation throughput: 304.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:51:56 [loggers.py:248] Engine 000: Avg prompt throughput: 36.1 tokens/s, Avg generation throughput: 219.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:57994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:52:06 [loggers.py:248] Engine 000: Avg prompt throughput: 238.5 tokens/s, Avg generation throughput: 176.6 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:38000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:52:16 [loggers.py:248] Engine 000: Avg prompt throughput: 274.2 tokens/s, Avg generation throughput: 261.8 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:51354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58056 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:52126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:52:26 [loggers.py:248] Engine 000: Avg prompt throughput: 97.1 tokens/s, Avg generation throughput: 219.6 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:52126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:52:36 [loggers.py:248] Engine 000: Avg prompt throughput: 94.5 tokens/s, Avg generation throughput: 180.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:52126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:52:46 [loggers.py:248] Engine 000: Avg prompt throughput: 106.3 tokens/s, Avg generation throughput: 143.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:42760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:52126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43278 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:52126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:40500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:52126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:45070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:45084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:52:56 [loggers.py:248] Engine 000: Avg prompt throughput: 230.9 tokens/s, Avg generation throughput: 187.0 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:45084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58040 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:58034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:45100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:53:06 [loggers.py:248] Engine 000: Avg prompt throughput: 122.7 tokens/s, Avg generation throughput: 248.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:53:16 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 103.4 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:53:26 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 16.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:53:36 [loggers.py:248] Engine 000: Avg prompt throughput: 321.3 tokens/s, Avg generation throughput: 125.4 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 7.4%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:53:46 [loggers.py:248] Engine 000: Avg prompt throughput: 1139.4 tokens/s, Avg generation throughput: 526.7 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.7%, Prefix cache hit rate: 26.7%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:53:56 [loggers.py:248] Engine 000: Avg prompt throughput: 1121.3 tokens/s, Avg generation throughput: 805.2 tokens/s, Running: 32 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.0%, Prefix cache hit rate: 39.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:54:06 [loggers.py:248] Engine 000: Avg prompt throughput: 612.6 tokens/s, Avg generation throughput: 864.9 tokens/s, Running: 41 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.3%, Prefix cache hit rate: 43.9%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:54:16 [loggers.py:248] Engine 000: Avg prompt throughput: 620.8 tokens/s, Avg generation throughput: 856.1 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.3%, Prefix cache hit rate: 47.8%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:54:26 [loggers.py:248] Engine 000: Avg prompt throughput: 922.2 tokens/s, Avg generation throughput: 816.9 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.7%, Prefix cache hit rate: 42.6%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55104 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:54:36 [loggers.py:248] Engine 000: Avg prompt throughput: 981.1 tokens/s, Avg generation throughput: 967.6 tokens/s, Running: 42 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.1%, Prefix cache hit rate: 38.2%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:54:46 [loggers.py:248] Engine 000: Avg prompt throughput: 481.1 tokens/s, Avg generation throughput: 802.9 tokens/s, Running: 32 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 36.4%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:54:56 [loggers.py:248] Engine 000: Avg prompt throughput: 966.1 tokens/s, Avg generation throughput: 930.3 tokens/s, Running: 50 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.5%, Prefix cache hit rate: 33.2%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:55:06 [loggers.py:248] Engine 000: Avg prompt throughput: 447.5 tokens/s, Avg generation throughput: 1072.4 tokens/s, Running: 36 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 31.9%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37716 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:55:16 [loggers.py:248] Engine 000: Avg prompt throughput: 809.3 tokens/s, Avg generation throughput: 813.6 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.0%, Prefix cache hit rate: 29.8%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:55:26 [loggers.py:248] Engine 000: Avg prompt throughput: 991.2 tokens/s, Avg generation throughput: 638.6 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.7%, Prefix cache hit rate: 27.5%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:55:36 [loggers.py:248] Engine 000: Avg prompt throughput: 492.4 tokens/s, Avg generation throughput: 661.6 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.9%, Prefix cache hit rate: 26.5%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:55:46 [loggers.py:248] Engine 000: Avg prompt throughput: 946.8 tokens/s, Avg generation throughput: 797.0 tokens/s, Running: 41 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.5%, Prefix cache hit rate: 25.0%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:55:56 [loggers.py:248] Engine 000: Avg prompt throughput: 737.3 tokens/s, Avg generation throughput: 746.0 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.3%, Prefix cache hit rate: 23.8%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:56:06 [loggers.py:248] Engine 000: Avg prompt throughput: 743.4 tokens/s, Avg generation throughput: 694.3 tokens/s, Running: 36 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 22.7%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:56:16 [loggers.py:248] Engine 000: Avg prompt throughput: 606.7 tokens/s, Avg generation throughput: 723.4 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.0%, Prefix cache hit rate: 21.8%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:54024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:56452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:56:26 [loggers.py:248] Engine 000: Avg prompt throughput: 976.4 tokens/s, Avg generation throughput: 808.8 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.9%, Prefix cache hit rate: 20.6%
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:55106 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:44356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:60366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:53950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:41630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO: 127.0.0.1:43302 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:56:36 [loggers.py:248] Engine 000: Avg prompt throughput: 641.7 tokens/s, Avg generation throughput: 867.9 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.4%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:56:46 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 391.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.3%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:56:56 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 143.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=3950)[0;0m INFO 12-09 18:57:00 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1209 18:57:00.612911629 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
+[0;36m(APIServer pid=3950)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=3950)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=3950)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
new file mode 100644
index 0000000..f16452d
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 410.0767225319996,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 2.438568065569649,
+ "tokens_per_second": 1795.5908237208996
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..208cee5
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-073f047f-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 190.90 \nTotal input tokens: 37841 \nTotal generated tokens: 38766 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 203.07 \nPeak output token throughput (tok/s): 413.00 \nPeak concurrent requests: 11.00 \nTotal Token throughput (tok/s): 401.29 \n---------------Time to First Token----------------\nMean TTFT (ms): 58.57 \nMedian TTFT (ms): 47.42 \nP99 TTFT (ms): 140.81 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 21.86 \nMedian TPOT (ms): 21.74 \nP99 TPOT (ms): 23.96 \n---------------Inter-token Latency----------------\nMean ITL (ms): 21.82 \nMedian ITL (ms): 21.46 \nP99 ITL (ms): 32.13 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..dd55e8d
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-2ef19409-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 196.59 \nTotal input tokens: 145810 \nTotal generated tokens: 151842 \nRequest throughput (req/s): 3.66 \nOutput token throughput (tok/s): 772.38 \nPeak output token throughput (tok/s): 1274.00 \nPeak concurrent requests: 43.00 \nTotal Token throughput (tok/s): 1514.08 \n---------------Time to First Token----------------\nMean TTFT (ms): 57.73 \nMedian TTFT (ms): 49.35 \nP99 TTFT (ms): 135.16 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 25.29 \nMedian TPOT (ms): 24.90 \nP99 TPOT (ms): 34.94 \n---------------Inter-token Latency----------------\nMean ITL (ms): 25.09 \nMedian ITL (ms): 23.53 \nP99 ITL (ms): 78.49 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log
new file mode 100644
index 0000000..2791aba
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_server.log
@@ -0,0 +1,1065 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:08 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:08 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 65536, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:12 [model.py:629] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:12 [model.py:1755] Using max model len 65536
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:12 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=22156)[0;0m INFO 12-09 20:11:17 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=65536, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=22156)[0;0m WARNING 12-09 20:11:17 [multiproc_executor.py:880] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+INFO 12-09 20:11:20 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:43039 backend=nccl
+INFO 12-09 20:11:20 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:43039 backend=nccl
+INFO 12-09 20:11:20 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-09 20:11:21 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-09 20:11:21 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:21 [gpu_model_runner.py:3544] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(Worker_TP1 pid=22239)[0;0m INFO 12-09 20:11:22 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:22 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=22238)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 9.33it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:00<00:00, 2.26it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:01<00:00, 1.82it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.80it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.98it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:25 [default_loader.py:308] Loading weights took 2.02 seconds
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:25 [gpu_model_runner.py:3626] Model loading took 7.5820 GiB memory and 3.455071 seconds
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:28 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/691e17755b/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:28 [backends.py:676] Dynamo bytecode transform time: 2.30 s
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:32 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP1 pid=22239)[0;0m INFO 12-09 20:11:32 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:41 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 11.76 s
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:41 [monitor.py:34] torch.compile takes 14.06 s in total
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:43 [gpu_worker.py:364] Available KV cache memory: 20.81 GiB
+[0;36m(EngineCore_DP0 pid=22156)[0;0m INFO 12-09 20:11:44 [kv_cache_utils.py:1287] GPU KV cache size: 340,928 tokens
+[0;36m(EngineCore_DP0 pid=22156)[0;0m INFO 12-09 20:11:44 [kv_cache_utils.py:1292] Maximum concurrency for 65,536 tokens per request: 5.20x
+[0;36m(Worker_TP0 pid=22238)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:05, 3.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:04, 3.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:04, 3.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:01<00:04, 3.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:01<00:03, 3.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:01<00:03, 3.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:01<00:03, 3.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:02<00:03, 3.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:02<00:02, 3.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:02<00:02, 3.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:03<00:02, 3.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:03<00:01, 3.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:03<00:01, 3.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:03<00:01, 3.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:04<00:01, 3.62it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:04<00:00, 3.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:04<00:00, 3.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:04<00:00, 3.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:05<00:00, 3.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:05<00:00, 3.64it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:02, 3.67it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:02, 3.71it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:02, 3.76it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:01<00:01, 3.78it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:01<00:01, 3.83it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:01<00:01, 3.74it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:01<00:01, 3.71it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:02<00:00, 3.74it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:02<00:00, 3.71it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:02<00:00, 3.72it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:02<00:00, 3.70it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:02<00:00, 3.73it/s]
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:11:53 [gpu_model_runner.py:4548] Graph capturing finished in 9 secs, took 0.22 GiB
+[0;36m(EngineCore_DP0 pid=22156)[0;0m INFO 12-09 20:11:53 [core.py:256] init engine (profile, create kv cache, warmup model) took 27.05 seconds
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=21994)[0;0m WARNING 12-09 20:11:55 [model.py:1581] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [serving_responses.py:197] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:11:55 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=21994)[0;0m INFO: Started server process [21994]
+[0;36m(APIServer pid=21994)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=21994)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:39104 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:12:15 [loggers.py:248] Engine 000: Avg prompt throughput: 41.7 tokens/s, Avg generation throughput: 54.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:12:25 [loggers.py:248] Engine 000: Avg prompt throughput: 135.3 tokens/s, Avg generation throughput: 149.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:12:35 [loggers.py:248] Engine 000: Avg prompt throughput: 183.3 tokens/s, Avg generation throughput: 220.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:12:45 [loggers.py:248] Engine 000: Avg prompt throughput: 278.0 tokens/s, Avg generation throughput: 134.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:12:55 [loggers.py:248] Engine 000: Avg prompt throughput: 191.1 tokens/s, Avg generation throughput: 192.6 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:13:05 [loggers.py:248] Engine 000: Avg prompt throughput: 365.3 tokens/s, Avg generation throughput: 235.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35024 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:13:15 [loggers.py:248] Engine 000: Avg prompt throughput: 366.2 tokens/s, Avg generation throughput: 217.0 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:13:25 [loggers.py:248] Engine 000: Avg prompt throughput: 151.0 tokens/s, Avg generation throughput: 187.7 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:13:35 [loggers.py:248] Engine 000: Avg prompt throughput: 373.9 tokens/s, Avg generation throughput: 256.1 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:13:45 [loggers.py:248] Engine 000: Avg prompt throughput: 223.3 tokens/s, Avg generation throughput: 341.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:36822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:13:55 [loggers.py:248] Engine 000: Avg prompt throughput: 276.0 tokens/s, Avg generation throughput: 227.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:14:05 [loggers.py:248] Engine 000: Avg prompt throughput: 36.1 tokens/s, Avg generation throughput: 194.9 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:14:15 [loggers.py:248] Engine 000: Avg prompt throughput: 238.5 tokens/s, Avg generation throughput: 200.5 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:14:25 [loggers.py:248] Engine 000: Avg prompt throughput: 274.2 tokens/s, Avg generation throughput: 250.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:44916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:60994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:35020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:14:35 [loggers.py:248] Engine 000: Avg prompt throughput: 97.1 tokens/s, Avg generation throughput: 273.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:14:45 [loggers.py:248] Engine 000: Avg prompt throughput: 94.5 tokens/s, Avg generation throughput: 85.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:55448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:14:55 [loggers.py:248] Engine 000: Avg prompt throughput: 106.3 tokens/s, Avg generation throughput: 177.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51998 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:51990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:48728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:55448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:15:05 [loggers.py:248] Engine 000: Avg prompt throughput: 230.9 tokens/s, Avg generation throughput: 203.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:45520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:55448 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:48732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:47610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:54524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:15:15 [loggers.py:248] Engine 000: Avg prompt throughput: 122.7 tokens/s, Avg generation throughput: 240.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:15:25 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 43.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:15:35 [loggers.py:248] Engine 000: Avg prompt throughput: 319.1 tokens/s, Avg generation throughput: 147.6 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 7.4%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:15:45 [loggers.py:248] Engine 000: Avg prompt throughput: 936.3 tokens/s, Avg generation throughput: 642.1 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 23.8%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:15:55 [loggers.py:248] Engine 000: Avg prompt throughput: 1214.3 tokens/s, Avg generation throughput: 885.6 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 38.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:16:05 [loggers.py:248] Engine 000: Avg prompt throughput: 674.7 tokens/s, Avg generation throughput: 954.9 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 43.6%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:16:15 [loggers.py:248] Engine 000: Avg prompt throughput: 632.1 tokens/s, Avg generation throughput: 752.8 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 48.0%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:16:25 [loggers.py:248] Engine 000: Avg prompt throughput: 848.8 tokens/s, Avg generation throughput: 937.2 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 43.2%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:16:35 [loggers.py:248] Engine 000: Avg prompt throughput: 1053.3 tokens/s, Avg generation throughput: 867.4 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 38.4%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:16:45 [loggers.py:248] Engine 000: Avg prompt throughput: 521.5 tokens/s, Avg generation throughput: 770.6 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 36.4%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:16:55 [loggers.py:248] Engine 000: Avg prompt throughput: 964.4 tokens/s, Avg generation throughput: 1003.2 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 33.2%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:17:05 [loggers.py:248] Engine 000: Avg prompt throughput: 447.3 tokens/s, Avg generation throughput: 1055.0 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 31.9%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:17:15 [loggers.py:248] Engine 000: Avg prompt throughput: 804.1 tokens/s, Avg generation throughput: 774.6 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 29.8%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:17:25 [loggers.py:248] Engine 000: Avg prompt throughput: 929.0 tokens/s, Avg generation throughput: 625.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 28.2%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:17:35 [loggers.py:248] Engine 000: Avg prompt throughput: 540.2 tokens/s, Avg generation throughput: 682.8 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 27.1%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:17:45 [loggers.py:248] Engine 000: Avg prompt throughput: 839.7 tokens/s, Avg generation throughput: 865.9 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 25.7%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:17:55 [loggers.py:248] Engine 000: Avg prompt throughput: 865.8 tokens/s, Avg generation throughput: 750.1 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 24.3%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:18:05 [loggers.py:248] Engine 000: Avg prompt throughput: 731.9 tokens/s, Avg generation throughput: 654.1 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 23.2%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:18:15 [loggers.py:248] Engine 000: Avg prompt throughput: 641.3 tokens/s, Avg generation throughput: 754.4 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 22.3%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:34590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:46042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:18:25 [loggers.py:248] Engine 000: Avg prompt throughput: 950.3 tokens/s, Avg generation throughput: 887.7 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 21.1%
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:52438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:40390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:41874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:49528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO: 127.0.0.1:38976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:18:35 [loggers.py:248] Engine 000: Avg prompt throughput: 667.8 tokens/s, Avg generation throughput: 849.4 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 20.3%
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:18:45 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 321.0 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 20.3%
+[0;36m(APIServer pid=21994)[0;0m INFO 12-09 20:18:47 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP0 pid=22238)[0;0m INFO 12-09 20:18:47 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(Worker_TP1 pid=22239)[0;0m INFO 12-09 20:18:47 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(APIServer pid=21994)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=21994)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=21994)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_throughput.json
new file mode 100644
index 0000000..e64892b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/meta-llama_Meta-Llama-3.1-8B-Instruct_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 269.04666597899995,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 3.7168273257028708,
+ "tokens_per_second": 2736.811464734795
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..bdf52ba
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-84df0cbe-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 217.42 \nTotal input tokens: 38756 \nTotal generated tokens: 39194 \nRequest throughput (req/s): 0.83 \nOutput token throughput (tok/s): 180.27 \nPeak output token throughput (tok/s): 290.00 \nPeak concurrent requests: 20.00 \nTotal Token throughput (tok/s): 358.52 \n---------------Time to First Token----------------\nMean TTFT (ms): 129.43 \nMedian TTFT (ms): 121.63 \nP99 TTFT (ms): 431.05 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 61.28 \nMedian TPOT (ms): 61.83 \nP99 TPOT (ms): 74.53 \n---------------Inter-token Latency----------------\nMean ITL (ms): 61.80 \nMedian ITL (ms): 59.99 \nP99 ITL (ms): 129.94 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..509d6d9
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-4ea75820-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 246.91 \nTotal input tokens: 145540 \nTotal generated tokens: 152007 \nRequest throughput (req/s): 2.92 \nOutput token throughput (tok/s): 615.63 \nPeak output token throughput (tok/s): 896.00 \nPeak concurrent requests: 109.00 \nTotal Token throughput (tok/s): 1205.06 \n---------------Time to First Token----------------\nMean TTFT (ms): 4078.97 \nMedian TTFT (ms): 3857.18 \nP99 TTFT (ms): 10112.65 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 78.49 \nMedian TPOT (ms): 79.02 \nP99 TPOT (ms): 93.10 \n---------------Inter-token Latency----------------\nMean ITL (ms): 78.52 \nMedian ITL (ms): 74.18 \nP99 ITL (ms): 163.52 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_server.log b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_server.log
new file mode 100644
index 0000000..b63a645
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_server.log
@@ -0,0 +1,1050 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:10:31 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:10:31 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.98, 'max_num_seqs': 64}
+[0;36m(APIServer pid=9049)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:10:35 [model.py:629] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=9049)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:05<00:10, 5.20s/it]
Parse safetensors files: 100%|██████████| 3/3 [00:05<00:00, 1.73s/it]
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:10:41 [model.py:1755] Using max model len 32768
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:10:41 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:10:41 [config.py:269] Overriding max cuda graph capture size to 1024 for performance.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:45 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:45 [parallel_state.py:1203] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://192.168.1.122:37735 backend=nccl
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:45 [parallel_state.py:1411] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:45 [gpu_model_runner.py:3544] Starting to load model openai/gpt-oss-20b...
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:45 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:45 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:45 [mxfp4.py:171] Using Triton backend
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:01, 1.38it/s]
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:01<00:00, 1.13it/s]
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.07it/s]
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.10it/s]
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:49 [default_loader.py:308] Loading weights took 2.79 seconds
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:49 [gpu_model_runner.py:3626] Model loading took 14.3066 GiB memory and 3.481873 seconds
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:51 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/fd3d592b37/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:51 [backends.py:676] Dynamo bytecode transform time: 1.70 s
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:10:54 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:11:13 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 20.60 s
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:11:13 [monitor.py:34] torch.compile takes 22.30 s in total
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:11:15 [gpu_worker.py:364] Available KV cache memory: 13.34 GiB
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:11:15 [kv_cache_utils.py:1287] GPU KV cache size: 291,392 tokens
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:11:15 [kv_cache_utils.py:1292] Maximum concurrency for 32,768 tokens per request: 16.67x
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 1%| | 1/83 [00:00<00:14, 5.68it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:13, 6.10it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 4%|▎ | 3/83 [00:00<00:13, 6.15it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:00<00:12, 6.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 6%|▌ | 5/83 [00:00<00:12, 6.38it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:00<00:11, 6.51it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 8%|▊ | 7/83 [00:01<00:11, 6.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:01<00:11, 6.62it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 9/83 [00:01<00:11, 6.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:01<00:10, 6.80it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 13%|█▎ | 11/83 [00:01<00:10, 6.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:01<00:10, 6.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 13/83 [00:01<00:09, 7.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:02<00:09, 7.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 15/83 [00:02<00:09, 7.18it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:02<00:09, 7.19it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 20%|██ | 17/83 [00:02<00:09, 7.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:02<00:08, 7.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 23%|██▎ | 19/83 [00:02<00:08, 7.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:02<00:08, 7.62it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 25%|██▌ | 21/83 [00:03<00:08, 7.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:03<00:07, 7.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 28%|██▊ | 23/83 [00:03<00:07, 7.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:03<00:07, 8.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 30%|███ | 25/83 [00:03<00:07, 8.18it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:03<00:06, 8.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 33%|███▎ | 27/83 [00:03<00:06, 8.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:03<00:06, 8.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 35%|███▍ | 29/83 [00:03<00:06, 8.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 31/83 [00:04<00:05, 9.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 40%|███▉ | 33/83 [00:04<00:05, 9.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 35/83 [00:04<00:04, 9.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▍ | 37/83 [00:04<00:04, 10.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 39/83 [00:04<00:04, 10.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 49%|████▉ | 41/83 [00:05<00:03, 10.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 52%|█████▏ | 43/83 [00:05<00:03, 11.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 54%|█████▍ | 45/83 [00:05<00:03, 11.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 57%|█████▋ | 47/83 [00:05<00:02, 12.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 59%|█████▉ | 49/83 [00:05<00:02, 13.13it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 61%|██████▏ | 51/83 [00:05<00:02, 13.80it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▍ | 53/83 [00:05<00:02, 14.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 66%|██████▋ | 55/83 [00:06<00:01, 14.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 69%|██████▊ | 57/83 [00:06<00:01, 15.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 71%|███████ | 59/83 [00:06<00:01, 15.70it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 61/83 [00:06<00:01, 16.15it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 76%|███████▌ | 63/83 [00:06<00:01, 16.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 78%|███████▊ | 65/83 [00:06<00:01, 17.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:06<00:00, 18.74it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 86%|████████▌ | 71/83 [00:06<00:00, 19.72it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:06<00:00, 20.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 93%|█████████▎| 77/83 [00:07<00:00, 22.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 96%|█████████▋| 80/83 [00:07<00:00, 23.69it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:07<00:00, 18.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:07<00:00, 11.13it/s]
+[0;36m(EngineCore_DP0 pid=9216)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 24.74it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 26.84it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 23.05it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 21.40it/s]
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:11:24 [gpu_model_runner.py:4548] Graph capturing finished in 9 secs, took 0.77 GiB
+[0;36m(EngineCore_DP0 pid=9216)[0;0m INFO 12-09 19:11:24 [core.py:256] init engine (profile, create kv cache, warmup model) took 34.56 seconds
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:25 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=9049)[0;0m WARNING 12-09 19:11:25 [serving_responses.py:218] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:28 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=9049)[0;0m INFO: Started server process [9049]
+[0;36m(APIServer pid=9049)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=9049)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56250 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:48 [loggers.py:248] Engine 000: Avg prompt throughput: 4.9 tokens/s, Avg generation throughput: 15.7 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:11:58 [loggers.py:248] Engine 000: Avg prompt throughput: 132.7 tokens/s, Avg generation throughput: 85.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:12:08 [loggers.py:248] Engine 000: Avg prompt throughput: 223.5 tokens/s, Avg generation throughput: 139.8 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:12:18 [loggers.py:248] Engine 000: Avg prompt throughput: 275.0 tokens/s, Avg generation throughput: 164.6 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:12:28 [loggers.py:248] Engine 000: Avg prompt throughput: 225.3 tokens/s, Avg generation throughput: 184.5 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:12:38 [loggers.py:248] Engine 000: Avg prompt throughput: 365.1 tokens/s, Avg generation throughput: 200.4 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:12:48 [loggers.py:248] Engine 000: Avg prompt throughput: 331.9 tokens/s, Avg generation throughput: 217.3 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:12:58 [loggers.py:248] Engine 000: Avg prompt throughput: 274.5 tokens/s, Avg generation throughput: 212.4 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:13:08 [loggers.py:248] Engine 000: Avg prompt throughput: 288.6 tokens/s, Avg generation throughput: 195.0 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:13:18 [loggers.py:248] Engine 000: Avg prompt throughput: 174.4 tokens/s, Avg generation throughput: 249.5 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:13:28 [loggers.py:248] Engine 000: Avg prompt throughput: 377.1 tokens/s, Avg generation throughput: 254.5 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:60064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:13:39 [loggers.py:248] Engine 000: Avg prompt throughput: 42.4 tokens/s, Avg generation throughput: 269.5 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:13:49 [loggers.py:248] Engine 000: Avg prompt throughput: 196.8 tokens/s, Avg generation throughput: 227.6 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:60064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:41488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:60064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:41488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:13:59 [loggers.py:248] Engine 000: Avg prompt throughput: 243.4 tokens/s, Avg generation throughput: 250.1 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:14:09 [loggers.py:248] Engine 000: Avg prompt throughput: 155.6 tokens/s, Avg generation throughput: 231.5 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:14:19 [loggers.py:248] Engine 000: Avg prompt throughput: 63.4 tokens/s, Avg generation throughput: 179.1 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:46742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:41488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:14:29 [loggers.py:248] Engine 000: Avg prompt throughput: 156.7 tokens/s, Avg generation throughput: 156.8 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:14:39 [loggers.py:248] Engine 000: Avg prompt throughput: 89.6 tokens/s, Avg generation throughput: 194.1 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:41488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:60064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:35208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:60064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:34320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:36142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:14:49 [loggers.py:248] Engine 000: Avg prompt throughput: 256.0 tokens/s, Avg generation throughput: 211.3 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:14:59 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 185.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:15:09 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 67.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:15:19 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 25.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:15:29 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.3 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:15:39 [loggers.py:248] Engine 000: Avg prompt throughput: 214.7 tokens/s, Avg generation throughput: 55.7 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 4.9%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:15:49 [loggers.py:248] Engine 000: Avg prompt throughput: 1016.7 tokens/s, Avg generation throughput: 319.2 tokens/s, Running: 35 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 23.2%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:15:59 [loggers.py:248] Engine 000: Avg prompt throughput: 1218.5 tokens/s, Avg generation throughput: 594.9 tokens/s, Running: 54 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 37.3%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:16:09 [loggers.py:248] Engine 000: Avg prompt throughput: 707.5 tokens/s, Avg generation throughput: 679.7 tokens/s, Running: 57 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 43.1%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:16:19 [loggers.py:248] Engine 000: Avg prompt throughput: 521.6 tokens/s, Avg generation throughput: 778.2 tokens/s, Running: 60 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 46.7%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:16:29 [loggers.py:248] Engine 000: Avg prompt throughput: 730.5 tokens/s, Avg generation throughput: 773.8 tokens/s, Running: 64 reqs, Waiting: 13 reqs, GPU KV cache usage: 5.7%, Prefix cache hit rate: 44.9%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:16:39 [loggers.py:248] Engine 000: Avg prompt throughput: 910.8 tokens/s, Avg generation throughput: 768.0 tokens/s, Running: 63 reqs, Waiting: 10 reqs, GPU KV cache usage: 5.6%, Prefix cache hit rate: 40.4%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:16:49 [loggers.py:248] Engine 000: Avg prompt throughput: 670.6 tokens/s, Avg generation throughput: 793.6 tokens/s, Running: 63 reqs, Waiting: 14 reqs, GPU KV cache usage: 5.6%, Prefix cache hit rate: 37.7%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:16:59 [loggers.py:248] Engine 000: Avg prompt throughput: 549.2 tokens/s, Avg generation throughput: 825.6 tokens/s, Running: 64 reqs, Waiting: 32 reqs, GPU KV cache usage: 5.4%, Prefix cache hit rate: 35.7%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:17:09 [loggers.py:248] Engine 000: Avg prompt throughput: 656.7 tokens/s, Avg generation throughput: 812.8 tokens/s, Running: 64 reqs, Waiting: 32 reqs, GPU KV cache usage: 4.7%, Prefix cache hit rate: 33.6%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:17:19 [loggers.py:248] Engine 000: Avg prompt throughput: 629.6 tokens/s, Avg generation throughput: 819.1 tokens/s, Running: 64 reqs, Waiting: 32 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 31.8%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:33280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:17:29 [loggers.py:248] Engine 000: Avg prompt throughput: 850.0 tokens/s, Avg generation throughput: 787.2 tokens/s, Running: 63 reqs, Waiting: 27 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 30.2%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:17:39 [loggers.py:248] Engine 000: Avg prompt throughput: 950.7 tokens/s, Avg generation throughput: 787.2 tokens/s, Running: 64 reqs, Waiting: 15 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 28.1%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:17:49 [loggers.py:248] Engine 000: Avg prompt throughput: 774.6 tokens/s, Avg generation throughput: 767.6 tokens/s, Running: 64 reqs, Waiting: 28 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 26.6%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:17:59 [loggers.py:248] Engine 000: Avg prompt throughput: 676.3 tokens/s, Avg generation throughput: 812.7 tokens/s, Running: 64 reqs, Waiting: 17 reqs, GPU KV cache usage: 4.9%, Prefix cache hit rate: 25.5%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39158 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:18:09 [loggers.py:248] Engine 000: Avg prompt throughput: 1118.8 tokens/s, Avg generation throughput: 755.2 tokens/s, Running: 64 reqs, Waiting: 6 reqs, GPU KV cache usage: 4.9%, Prefix cache hit rate: 23.7%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47592 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55660 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:59172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:37936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:18:19 [loggers.py:248] Engine 000: Avg prompt throughput: 585.3 tokens/s, Avg generation throughput: 787.1 tokens/s, Running: 64 reqs, Waiting: 14 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 22.9%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53646 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:18:29 [loggers.py:248] Engine 000: Avg prompt throughput: 600.5 tokens/s, Avg generation throughput: 799.9 tokens/s, Running: 64 reqs, Waiting: 27 reqs, GPU KV cache usage: 5.5%, Prefix cache hit rate: 22.1%
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55640 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:53614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:51270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57858 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:57934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:39134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:56032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:58872 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:55738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:18:39 [loggers.py:248] Engine 000: Avg prompt throughput: 921.1 tokens/s, Avg generation throughput: 755.1 tokens/s, Running: 64 reqs, Waiting: 20 reqs, GPU KV cache usage: 5.0%, Prefix cache hit rate: 21.0%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:18:49 [loggers.py:248] Engine 000: Avg prompt throughput: 250.5 tokens/s, Avg generation throughput: 811.3 tokens/s, Running: 54 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:18:59 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 486.3 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:19:09 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 244.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:19:19 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 137.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:19:29 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 29.2 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:19:39 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 23.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=9049)[0;0m INFO 12-09 19:19:42 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1209 19:19:42.316721556 ProcessGroupNCCL.cpp:1553] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
+[0;36m(APIServer pid=9049)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=9049)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=9049)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_throughput.json b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_throughput.json
new file mode 100644
index 0000000..f89973b
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 674.5005297629996,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 1.4825785242175744,
+ "tokens_per_second": 1095.31715306375
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps1.0_latency.json b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps1.0_latency.json
new file mode 100644
index 0000000..7e3f66c
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-7a645b9b-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 199.44 \nTotal input tokens: 38756 \nTotal generated tokens: 38779 \nRequest throughput (req/s): 0.90 \nOutput token throughput (tok/s): 194.44 \nPeak output token throughput (tok/s): 395.00 \nPeak concurrent requests: 14.00 \nTotal Token throughput (tok/s): 388.76 \n---------------Time to First Token----------------\nMean TTFT (ms): 75.40 \nMedian TTFT (ms): 67.78 \nP99 TTFT (ms): 192.76 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 28.60 \nMedian TPOT (ms): 28.16 \nP99 TPOT (ms): 41.84 \n---------------Inter-token Latency----------------\nMean ITL (ms): 29.20 \nMedian ITL (ms): 25.34 \nP99 ITL (ms): 71.24 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps4.0_latency.json b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps4.0_latency.json
new file mode 100644
index 0000000..67a6520
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": false,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, tokenizer_mode='auto', use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-c0304eb1-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, common_prefix_len=None, served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nERROR 12-09 20:35:02 [repo_utils.py:65] Error retrieving file list: 502 Server Error: Bad Gateway for url: https://huggingface.co/api/models/openai/gpt-oss-20b/tree/main?recursive=True&expand=False, retrying 1 of 2\nERROR 12-09 20:35:04 [repo_utils.py:63] Error retrieving file list: 502 Server Error: Bad Gateway for url: https://huggingface.co/api/models/openai/gpt-oss-20b/tree/main?recursive=True&expand=False\nERROR 12-09 20:35:04 [repo_utils.py:128] Error retrieving file list. Please ensure your `model_name_or_path``repo_type`, `token` and `revision` arguments are correctly set. Returning an empty list.\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_server.log b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_server.log
new file mode 100644
index 0000000..f08bdec
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_server.log
@@ -0,0 +1,328 @@
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:30:09 [api_server.py:1351] vLLM API server version 0.11.2.dev690+g67475a6e8.d20251209
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:30:09 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 32768, 'tensor_parallel_size': 2, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=28665)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:30:13 [model.py:629] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=28665)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 4.52it/s]
Parse safetensors files: 67%|██████▋ | 2/3 [00:00<00:00, 4.96it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 7.33it/s]
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:30:14 [model.py:1755] Using max model len 32768
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:30:14 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:30:14 [config.py:269] Overriding max cuda graph capture size to 1024 for performance.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+[0;36m(EngineCore_DP0 pid=28832)[0;0m INFO 12-09 20:30:18 [core.py:93] Initializing a V1 LLM engine (v0.11.2.dev690+g67475a6e8.d20251209) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=2, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=True, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01, cudagraph_metrics=False, enable_layerwise_nvtx_tracing=False), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'compile_ranges_split_points': [2048], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': , 'evaluate_guards': False}, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=28832)[0;0m WARNING 12-09 20:30:18 [multiproc_executor.py:880] Reducing Torch parallelism from 24 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+/opt/venv/lib64/python3.13/site-packages/torch/library.py:357: UserWarning: Warning only once for all operators, other operators may also be overridden.
+ Overriding a previously registered kernel for the same operator and the same dispatch key
+ operator: flash_attn::_flash_attn_backward(Tensor dout, Tensor q, Tensor k, Tensor v, Tensor out, Tensor softmax_lse, Tensor(a6!)? dq, Tensor(a7!)? dk, Tensor(a8!)? dv, float dropout_p, float softmax_scale, bool causal, SymInt window_size_left, SymInt window_size_right, float softcap, Tensor? alibi_slopes, bool deterministic, Tensor? rng_state=None) -> Tensor
+ registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926
+ dispatch key: ADInplaceOrView
+ previous kernel: no debug info
+ new kernel: registered at /opt/venv/lib64/python3.13/site-packages/torch/_library/custom_ops.py:926 (Triggered internally at /__w/TheRock/TheRock/external-builds/pytorch/pytorch/aten/src/ATen/core/dispatch/OperatorEntry.cpp:208.)
+ self.m.impl(
+INFO 12-09 20:30:22 [parallel_state.py:1203] world_size=2 rank=1 local_rank=1 distributed_init_method=tcp://127.0.0.1:58745 backend=nccl
+INFO 12-09 20:30:22 [parallel_state.py:1203] world_size=2 rank=0 local_rank=0 distributed_init_method=tcp://127.0.0.1:58745 backend=nccl
+INFO 12-09 20:30:22 [pynccl.py:111] vLLM is using nccl==2.27.3
+INFO 12-09 20:30:22 [parallel_state.py:1411] rank 0 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+INFO 12-09 20:30:22 [parallel_state.py:1411] rank 1 in world size 2 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 1, EP rank 1
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:23 [gpu_model_runner.py:3544] Starting to load model openai/gpt-oss-20b...
+[0;36m(Worker_TP1 pid=28915)[0;0m INFO 12-09 20:30:23 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP1 pid=28915)[0;0m INFO 12-09 20:30:23 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP1 pid=28915)[0;0m INFO 12-09 20:30:23 [mxfp4.py:171] Using Triton backend
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:23 [rocm.py:320] Using Triton Attention backend on V1 engine.
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:23 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:23 [mxfp4.py:171] Using Triton backend
+[0;36m(Worker_TP0 pid=28914)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(Worker_TP0 pid=28914)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:00, 2.43it/s]
+[0;36m(Worker_TP0 pid=28914)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:00<00:00, 1.99it/s]
+[0;36m(Worker_TP0 pid=28914)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:01<00:00, 1.91it/s]
+[0;36m(Worker_TP0 pid=28914)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:01<00:00, 1.97it/s]
+[0;36m(Worker_TP0 pid=28914)[0;0m
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:26 [default_loader.py:308] Loading weights took 1.55 seconds
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:26 [gpu_model_runner.py:3626] Model loading took 7.4551 GiB memory and 2.653995 seconds
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:28 [backends.py:616] Using cache directory: /home/kyuz0/.cache/vllm/torch_compile_cache/c0e9e7ea2d/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:28 [backends.py:676] Dynamo bytecode transform time: 1.88 s
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:30 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP1 pid=28915)[0;0m INFO 12-09 20:30:30 [backends.py:243] Cache the graph of compile range (1, 2048) for later use
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:49 [backends.py:260] Compiling a graph for compile range (1, 2048) takes 19.47 s
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:49 [monitor.py:34] torch.compile takes 21.36 s in total
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:30:51 [gpu_worker.py:364] Available KV cache memory: 19.83 GiB
+[0;36m(EngineCore_DP0 pid=28832)[0;0m INFO 12-09 20:30:51 [kv_cache_utils.py:1287] GPU KV cache size: 864,576 tokens
+[0;36m(EngineCore_DP0 pid=28832)[0;0m INFO 12-09 20:30:51 [kv_cache_utils.py:1292] Maximum concurrency for 32,768 tokens per request: 49.46x
+[0;36m(Worker_TP0 pid=28914)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 1%| | 1/83 [00:00<00:25, 3.21it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:24, 3.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 4%|▎ | 3/83 [00:00<00:24, 3.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:01<00:23, 3.29it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 6%|▌ | 5/83 [00:01<00:23, 3.29it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:01<00:23, 3.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 8%|▊ | 7/83 [00:02<00:22, 3.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:02<00:22, 3.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 9/83 [00:02<00:22, 3.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:03<00:21, 3.38it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 13%|█▎ | 11/83 [00:03<00:21, 3.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:03<00:21, 3.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 13/83 [00:03<00:20, 3.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:04<00:20, 3.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 15/83 [00:04<00:19, 3.41it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:04<00:19, 3.41it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 20%|██ | 17/83 [00:05<00:19, 3.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:05<00:18, 3.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 23%|██▎ | 19/83 [00:05<00:18, 3.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:05<00:18, 3.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 25%|██▌ | 21/83 [00:06<00:18, 3.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:06<00:17, 3.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 28%|██▊ | 23/83 [00:06<00:17, 3.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:07<00:16, 3.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 30%|███ | 25/83 [00:07<00:16, 3.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:07<00:16, 3.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 33%|███▎ | 27/83 [00:07<00:16, 3.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:08<00:16, 3.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 35%|███▍ | 29/83 [00:08<00:15, 3.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▌ | 30/83 [00:08<00:15, 3.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 31/83 [00:09<00:14, 3.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 39%|███▊ | 32/83 [00:09<00:14, 3.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 40%|███▉ | 33/83 [00:09<00:13, 3.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 41%|████ | 34/83 [00:09<00:13, 3.58it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 35/83 [00:10<00:13, 3.58it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 43%|████▎ | 36/83 [00:10<00:13, 3.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▍ | 37/83 [00:10<00:12, 3.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 46%|████▌ | 38/83 [00:11<00:12, 3.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 39/83 [00:11<00:12, 3.60it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 48%|████▊ | 40/83 [00:11<00:12, 3.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 49%|████▉ | 41/83 [00:11<00:11, 3.51it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 51%|█████ | 42/83 [00:12<00:11, 3.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 52%|█████▏ | 43/83 [00:12<00:11, 3.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 44/83 [00:12<00:10, 3.58it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 54%|█████▍ | 45/83 [00:12<00:10, 3.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▌ | 46/83 [00:13<00:10, 3.70it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 57%|█████▋ | 47/83 [00:13<00:09, 3.79it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 48/83 [00:13<00:08, 3.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 59%|█████▉ | 49/83 [00:13<00:08, 3.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 60%|██████ | 50/83 [00:14<00:08, 3.88it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 61%|██████▏ | 51/83 [00:14<00:08, 3.90it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 52/83 [00:14<00:07, 4.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▍ | 53/83 [00:14<00:07, 4.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 65%|██████▌ | 54/83 [00:15<00:07, 4.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 66%|██████▋ | 55/83 [00:15<00:06, 4.10it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 67%|██████▋ | 56/83 [00:15<00:06, 4.14it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 69%|██████▊ | 57/83 [00:15<00:06, 4.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 70%|██████▉ | 58/83 [00:16<00:06, 4.15it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 71%|███████ | 59/83 [00:16<00:05, 4.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 72%|███████▏ | 60/83 [00:16<00:05, 4.22it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 61/83 [00:16<00:05, 4.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 75%|███████▍ | 62/83 [00:17<00:04, 4.23it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 76%|███████▌ | 63/83 [00:17<00:04, 4.21it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 77%|███████▋ | 64/83 [00:17<00:04, 4.14it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 78%|███████▊ | 65/83 [00:17<00:04, 4.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 80%|███████▉ | 66/83 [00:18<00:04, 4.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 81%|████████ | 67/83 [00:18<00:03, 4.11it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:18<00:03, 4.13it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 83%|████████▎ | 69/83 [00:18<00:03, 4.14it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 70/83 [00:19<00:03, 4.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 86%|████████▌ | 71/83 [00:19<00:02, 4.22it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 87%|████████▋ | 72/83 [00:19<00:02, 4.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 88%|████████▊ | 73/83 [00:19<00:02, 4.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:19<00:02, 4.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 90%|█████████ | 75/83 [00:20<00:01, 4.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 92%|█████████▏| 76/83 [00:20<00:01, 4.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 93%|█████████▎| 77/83 [00:20<00:01, 4.29it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 94%|█████████▍| 78/83 [00:20<00:01, 4.29it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▌| 79/83 [00:21<00:00, 4.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 96%|█████████▋| 80/83 [00:21<00:00, 4.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 98%|█████████▊| 81/83 [00:21<00:00, 4.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 99%|█████████▉| 82/83 [00:21<00:00, 4.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:22<00:00, 4.19it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:22<00:00, 3.76it/s]
+[0;36m(Worker_TP0 pid=28914)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:02, 4.31it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:02, 4.44it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:01, 4.45it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:01, 4.40it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:01<00:01, 4.38it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:01<00:01, 4.34it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:01<00:00, 4.14it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:01<00:00, 4.12it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:02<00:00, 4.15it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:02<00:00, 4.18it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:02<00:00, 4.20it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:02<00:00, 4.25it/s]
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:31:17 [gpu_model_runner.py:4548] Graph capturing finished in 25 secs, took 1.36 GiB
+[0;36m(EngineCore_DP0 pid=28832)[0;0m INFO 12-09 20:31:17 [core.py:256] init engine (profile, create kv cache, warmup model) took 50.51 seconds
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [api_server.py:1099] Supported tasks: ['generate']
+[0;36m(APIServer pid=28665)[0;0m WARNING 12-09 20:31:21 [serving_responses.py:218] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [api_server.py:1425] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:21 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=28665)[0;0m INFO: Started server process [28665]
+[0;36m(APIServer pid=28665)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=28665)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:56386 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:41 [loggers.py:248] Engine 000: Avg prompt throughput: 2.4 tokens/s, Avg generation throughput: 16.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:31:51 [loggers.py:248] Engine 000: Avg prompt throughput: 135.2 tokens/s, Avg generation throughput: 102.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:32:01 [loggers.py:248] Engine 000: Avg prompt throughput: 223.5 tokens/s, Avg generation throughput: 205.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:32:11 [loggers.py:248] Engine 000: Avg prompt throughput: 274.2 tokens/s, Avg generation throughput: 194.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:59736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:59736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:32:21 [loggers.py:248] Engine 000: Avg prompt throughput: 224.5 tokens/s, Avg generation throughput: 180.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:59736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:32:31 [loggers.py:248] Engine 000: Avg prompt throughput: 366.7 tokens/s, Avg generation throughput: 205.3 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:59736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:32:41 [loggers.py:248] Engine 000: Avg prompt throughput: 313.3 tokens/s, Avg generation throughput: 209.2 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:32:51 [loggers.py:248] Engine 000: Avg prompt throughput: 293.0 tokens/s, Avg generation throughput: 223.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:33:01 [loggers.py:248] Engine 000: Avg prompt throughput: 288.6 tokens/s, Avg generation throughput: 196.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:45572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:51500 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:33:11 [loggers.py:248] Engine 000: Avg prompt throughput: 174.4 tokens/s, Avg generation throughput: 331.5 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:43260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:43260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:33:21 [loggers.py:248] Engine 000: Avg prompt throughput: 377.1 tokens/s, Avg generation throughput: 322.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:33:31 [loggers.py:248] Engine 000: Avg prompt throughput: 35.5 tokens/s, Avg generation throughput: 159.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:33:41 [loggers.py:248] Engine 000: Avg prompt throughput: 203.7 tokens/s, Avg generation throughput: 224.5 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:43772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:43772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:33:51 [loggers.py:248] Engine 000: Avg prompt throughput: 242.5 tokens/s, Avg generation throughput: 227.1 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:43772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:43772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:34:01 [loggers.py:248] Engine 000: Avg prompt throughput: 156.5 tokens/s, Avg generation throughput: 245.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:43772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:34:11 [loggers.py:248] Engine 000: Avg prompt throughput: 33.7 tokens/s, Avg generation throughput: 112.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:34:21 [loggers.py:248] Engine 000: Avg prompt throughput: 120.5 tokens/s, Avg generation throughput: 116.4 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41332 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:34:31 [loggers.py:248] Engine 000: Avg prompt throughput: 155.5 tokens/s, Avg generation throughput: 233.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:41876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO: 127.0.0.1:40476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:34:41 [loggers.py:248] Engine 000: Avg prompt throughput: 256.0 tokens/s, Avg generation throughput: 265.8 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:34:51 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 85.0 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:35:01 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 31.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:35:11 [loggers.py:248] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=28665)[0;0m INFO 12-09 20:35:29 [launcher.py:110] Shutting down FastAPI HTTP server.
+[0;36m(Worker_TP1 pid=28915)[0;0m INFO 12-09 20:35:29 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(Worker_TP0 pid=28914)[0;0m INFO 12-09 20:35:29 [multiproc_executor.py:709] Parent process exited, terminating worker
+[0;36m(APIServer pid=28665)[0;0m INFO: Shutting down
+[0;36m(APIServer pid=28665)[0;0m INFO: Waiting for application shutdown.
+[0;36m(APIServer pid=28665)[0;0m INFO: Application shutdown complete.
diff --git a/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_throughput.json b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_throughput.json
new file mode 100644
index 0000000..6c78218
--- /dev/null
+++ b/benchmarks/benchmark_results_amd-r9700/openai_gpt-oss-20b_tp2_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 393.24469268199937,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 2.542945953522781,
+ "tokens_per_second": 1878.7081268950026
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Llama-3.1-8B-Instruct-FP8-block_tp1_server.log b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Llama-3.1-8B-Instruct-FP8-block_tp1_server.log
new file mode 100644
index 0000000..fe392f1
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Llama-3.1-8B-Instruct-FP8-block_tp1_server.log
@@ -0,0 +1,95 @@
+WARNING 12-10 09:53:36 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=4288)[0;0m INFO 12-10 09:53:36 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=4288)[0;0m INFO 12-10 09:53:36 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Llama-3.1-8B-Instruct-FP8-block', 'host': '127.0.0.1', 'model': 'RedHatAI/Llama-3.1-8B-Instruct-FP8-block', 'trust_remote_code': True, 'max_model_len': 65536, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=4288)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=4288)[0;0m INFO 12-10 09:53:47 [model.py:637] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=4288)[0;0m INFO 12-10 09:53:47 [model.py:1750] Using max model len 65536
+[0;36m(APIServer pid=4288)[0;0m INFO 12-10 09:53:47 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=4288)[0;0m Traceback (most recent call last):
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/huggingface_hub/utils/_http.py", line 409, in hf_raise_for_status
+[0;36m(APIServer pid=4288)[0;0m response.raise_for_status()
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/requests/models.py", line 1026, in raise_for_status
+[0;36m(APIServer pid=4288)[0;0m raise HTTPError(http_error_msg, response=self)
+[0;36m(APIServer pid=4288)[0;0m requests.exceptions.HTTPError: 503 Server Error: Service Temporarily Unavailable for url: https://huggingface.co/api/models/RedHatAI/Llama-3.1-8B-Instruct-FP8-block
+[0;36m(APIServer pid=4288)[0;0m
+[0;36m(APIServer pid=4288)[0;0m The above exception was the direct cause of the following exception:
+[0;36m(APIServer pid=4288)[0;0m
+[0;36m(APIServer pid=4288)[0;0m Traceback (most recent call last):
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/bin/vllm", line 7, in
+[0;36m(APIServer pid=4288)[0;0m sys.exit(main())
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/entrypoints/cli/main.py", line 73, in main
+[0;36m(APIServer pid=4288)[0;0m args.dispatch_function(args)
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/entrypoints/cli/serve.py", line 60, in cmd
+[0;36m(APIServer pid=4288)[0;0m uvloop.run(run_server(args))
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/uvloop/__init__.py", line 96, in run
+[0;36m(APIServer pid=4288)[0;0m return __asyncio.run(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/asyncio/runners.py", line 195, in run
+[0;36m(APIServer pid=4288)[0;0m return runner.run(main)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/asyncio/runners.py", line 118, in run
+[0;36m(APIServer pid=4288)[0;0m return self._loop.run_until_complete(task)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/uvloop/__init__.py", line 48, in wrapper
+[0;36m(APIServer pid=4288)[0;0m return await main
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/entrypoints/openai/api_server.py", line 1819, in run_server
+[0;36m(APIServer pid=4288)[0;0m await run_server_worker(listen_address, sock, args, **uvicorn_kwargs)
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/entrypoints/openai/api_server.py", line 1838, in run_server_worker
+[0;36m(APIServer pid=4288)[0;0m async with build_async_engine_client(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/contextlib.py", line 210, in __aenter__
+[0;36m(APIServer pid=4288)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/entrypoints/openai/api_server.py", line 183, in build_async_engine_client
+[0;36m(APIServer pid=4288)[0;0m async with build_async_engine_client_from_engine_args(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/contextlib.py", line 210, in __aenter__
+[0;36m(APIServer pid=4288)[0;0m return await anext(self.gen)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/entrypoints/openai/api_server.py", line 224, in build_async_engine_client_from_engine_args
+[0;36m(APIServer pid=4288)[0;0m async_llm = AsyncLLM.from_vllm_config(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/v1/engine/async_llm.py", line 223, in from_vllm_config
+[0;36m(APIServer pid=4288)[0;0m return cls(
+[0;36m(APIServer pid=4288)[0;0m ^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/v1/engine/async_llm.py", line 114, in __init__
+[0;36m(APIServer pid=4288)[0;0m tokenizer = init_tokenizer_from_config(self.model_config)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/tokenizers/registry.py", line 227, in init_tokenizer_from_config
+[0;36m(APIServer pid=4288)[0;0m return get_tokenizer(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/tokenizers/registry.py", line 191, in get_tokenizer
+[0;36m(APIServer pid=4288)[0;0m tokenizer = TokenizerRegistry.get_tokenizer(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/tokenizers/registry.py", line 86, in get_tokenizer
+[0;36m(APIServer pid=4288)[0;0m return item.from_pretrained(*args, **kwargs)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/vllm/tokenizers/hf.py", line 84, in from_pretrained
+[0;36m(APIServer pid=4288)[0;0m tokenizer = AutoTokenizer.from_pretrained(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/transformers/models/auto/tokenization_auto.py", line 1156, in from_pretrained
+[0;36m(APIServer pid=4288)[0;0m return tokenizer_class.from_pretrained(pretrained_model_name_or_path, *inputs, **kwargs)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/transformers/tokenization_utils_base.py", line 2113, in from_pretrained
+[0;36m(APIServer pid=4288)[0;0m return cls._from_pretrained(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/transformers/tokenization_utils_base.py", line 2395, in _from_pretrained
+[0;36m(APIServer pid=4288)[0;0m tokenizer = cls._patch_mistral_regex(
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/transformers/tokenization_utils_base.py", line 2438, in _patch_mistral_regex
+[0;36m(APIServer pid=4288)[0;0m if _is_local or is_base_mistral(pretrained_model_name_or_path):
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/transformers/tokenization_utils_base.py", line 2432, in is_base_mistral
+[0;36m(APIServer pid=4288)[0;0m model = model_info(model_id)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/huggingface_hub/utils/_validators.py", line 114, in _inner_fn
+[0;36m(APIServer pid=4288)[0;0m return fn(*args, **kwargs)
+[0;36m(APIServer pid=4288)[0;0m ^^^^^^^^^^^^^^^^^^^
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/huggingface_hub/hf_api.py", line 2638, in model_info
+[0;36m(APIServer pid=4288)[0;0m hf_raise_for_status(r)
+[0;36m(APIServer pid=4288)[0;0m File "/venv/main/lib/python3.12/site-packages/huggingface_hub/utils/_http.py", line 482, in hf_raise_for_status
+[0;36m(APIServer pid=4288)[0;0m raise _format(HfHubHTTPError, str(e), response) from e
+[0;36m(APIServer pid=4288)[0;0m huggingface_hub.errors.HfHubHTTPError: 503 Server Error: Service Temporarily Unavailable for url: https://huggingface.co/api/models/RedHatAI/Llama-3.1-8B-Instruct-FP8-block
diff --git a/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..1acd84f
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-f177f1fc-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 191.24 \nTotal input tokens: 38358 \nTotal generated tokens: 40296 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 210.71 \nPeak output token throughput (tok/s): 430.00 \nPeak concurrent requests: 13.00 \nTotal Token throughput (tok/s): 411.29 \n---------------Time to First Token----------------\nMean TTFT (ms): 145.79 \nMedian TTFT (ms): 110.27 \nP99 TTFT (ms): 399.40 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 23.46 \nMedian TPOT (ms): 22.67 \nP99 TPOT (ms): 36.03 \n---------------Inter-token Latency----------------\nMean ITL (ms): 23.37 \nMedian ITL (ms): 21.34 \nP99 ITL (ms): 84.18 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..de4d720
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-31b9a516-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 215.08 \nTotal input tokens: 146694 \nTotal generated tokens: 155647 \nRequest throughput (req/s): 3.35 \nOutput token throughput (tok/s): 723.67 \nPeak output token throughput (tok/s): 1208.00 \nPeak concurrent requests: 113.00 \nTotal Token throughput (tok/s): 1405.72 \n---------------Time to First Token----------------\nMean TTFT (ms): 11329.86 \nMedian TTFT (ms): 14049.92 \nP99 TTFT (ms): 20450.77 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 40.85 \nMedian TPOT (ms): 39.82 \nP99 TPOT (ms): 83.14 \n---------------Inter-token Latency----------------\nMean ITL (ms): 40.32 \nMedian ITL (ms): 26.70 \nP99 ITL (ms): 338.67 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..96e14f0
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1027 @@
+WARNING 12-10 09:23:01 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:23:01 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:23:01 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 4096, 'gpu_memory_utilization': 0.86, 'max_num_seqs': 32}
+[0;36m(APIServer pid=3017)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:23:09 [model.py:637] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:23:09 [model.py:1750] Using max model len 4096
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:23:09 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:18 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=4096, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 64, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:19 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:36297 backend=nccl
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:19 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:19 [gpu_model_runner.py:3467] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:19 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 3.56it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.52it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.18it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.11it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.23it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:24 [default_loader.py:308] Loading weights took 3.34 seconds
+[0;36m(EngineCore_DP0 pid=3057)[0;0m WARNING 12-10 09:23:24 [marlin_utils_fp8.py:98] Your GPU does not have native support for FP8 computation but FP8 quantization is being used. Weight-only FP8 compression will be used leveraging the Marlin kernel. This may degrade performance for compute-heavy workloads.
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:24 [gpu_model_runner.py:3549] Model loading took 15.3291 GiB memory and 4.892989 seconds
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:37 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/882f1c097b/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:37 [backends.py:715] Dynamo bytecode transform time: 12.64 s
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:38 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:52 [backends.py:288] Compiling a graph for dynamic shape takes 14.28 s
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:55 [monitor.py:34] torch.compile takes 26.92 s in total
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:57 [gpu_worker.py:359] Available KV cache memory: 4.64 GiB
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:57 [kv_cache_utils.py:1286] GPU KV cache size: 30,384 tokens
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:57 [kv_cache_utils.py:1291] Maximum concurrency for 4,096 tokens per request: 7.42x
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 2/11 [00:00<00:00, 15.75it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▋ | 4/11 [00:00<00:00, 16.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▍ | 6/11 [00:00<00:00, 16.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 73%|███████▎ | 8/11 [00:00<00:00, 16.78it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 91%|█████████ | 10/11 [00:00<00:00, 16.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:00<00:00, 16.75it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 29%|██▊ | 2/7 [00:00<00:00, 13.85it/s]
Capturing CUDA graphs (decode, FULL): 57%|█████▋ | 4/7 [00:00<00:00, 15.06it/s]
Capturing CUDA graphs (decode, FULL): 86%|████████▌ | 6/7 [00:00<00:00, 15.42it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:00<00:00, 15.38it/s]
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:59 [gpu_model_runner.py:4466] Graph capturing finished in 2 secs, took 0.14 GiB
+[0;36m(EngineCore_DP0 pid=3057)[0;0m INFO 12-10 09:23:59 [core.py:254] init engine (profile, create kv cache, warmup model) took 34.60 seconds
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=3017)[0;0m WARNING 12-10 09:24:01 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:01 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=3017)[0;0m INFO: Started server process [3017]
+[0;36m(APIServer pid=3017)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=3017)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60512 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:21 [loggers.py:236] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 3.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:31 [loggers.py:236] Engine 000: Avg prompt throughput: 82.8 tokens/s, Avg generation throughput: 114.4 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:41 [loggers.py:236] Engine 000: Avg prompt throughput: 239.0 tokens/s, Avg generation throughput: 155.7 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41338 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:24:51 [loggers.py:236] Engine 000: Avg prompt throughput: 111.0 tokens/s, Avg generation throughput: 200.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:25:01 [loggers.py:236] Engine 000: Avg prompt throughput: 326.7 tokens/s, Avg generation throughput: 170.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:25:11 [loggers.py:236] Engine 000: Avg prompt throughput: 228.1 tokens/s, Avg generation throughput: 202.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39926 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57538 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:25:21 [loggers.py:236] Engine 000: Avg prompt throughput: 276.5 tokens/s, Avg generation throughput: 218.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:43808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:25:31 [loggers.py:236] Engine 000: Avg prompt throughput: 463.0 tokens/s, Avg generation throughput: 259.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:25:41 [loggers.py:236] Engine 000: Avg prompt throughput: 268.5 tokens/s, Avg generation throughput: 141.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:25:51 [loggers.py:236] Engine 000: Avg prompt throughput: 185.4 tokens/s, Avg generation throughput: 341.8 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:41370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:26:01 [loggers.py:236] Engine 000: Avg prompt throughput: 368.5 tokens/s, Avg generation throughput: 339.9 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:26:11 [loggers.py:236] Engine 000: Avg prompt throughput: 87.1 tokens/s, Avg generation throughput: 222.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:26:21 [loggers.py:236] Engine 000: Avg prompt throughput: 190.0 tokens/s, Avg generation throughput: 199.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:56162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:26:31 [loggers.py:236] Engine 000: Avg prompt throughput: 253.5 tokens/s, Avg generation throughput: 203.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:57528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:49932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:26:41 [loggers.py:236] Engine 000: Avg prompt throughput: 79.6 tokens/s, Avg generation throughput: 317.7 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:60604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:26:51 [loggers.py:236] Engine 000: Avg prompt throughput: 130.1 tokens/s, Avg generation throughput: 164.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:27:01 [loggers.py:236] Engine 000: Avg prompt throughput: 81.0 tokens/s, Avg generation throughput: 91.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:27:11 [loggers.py:236] Engine 000: Avg prompt throughput: 139.6 tokens/s, Avg generation throughput: 240.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44422 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:27:21 [loggers.py:236] Engine 000: Avg prompt throughput: 275.0 tokens/s, Avg generation throughput: 214.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:51264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:27:31 [loggers.py:236] Engine 000: Avg prompt throughput: 50.4 tokens/s, Avg generation throughput: 215.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:27:41 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 23.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:27:51 [loggers.py:236] Engine 000: Avg prompt throughput: 8.3 tokens/s, Avg generation throughput: 21.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:28:01 [loggers.py:236] Engine 000: Avg prompt throughput: 851.6 tokens/s, Avg generation throughput: 340.8 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:28:11 [loggers.py:236] Engine 000: Avg prompt throughput: 1107.8 tokens/s, Avg generation throughput: 558.2 tokens/s, Running: 32 reqs, Waiting: 12 reqs, GPU KV cache usage: 43.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:28:21 [loggers.py:236] Engine 000: Avg prompt throughput: 652.5 tokens/s, Avg generation throughput: 784.0 tokens/s, Running: 31 reqs, Waiting: 13 reqs, GPU KV cache usage: 41.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:28:31 [loggers.py:236] Engine 000: Avg prompt throughput: 540.5 tokens/s, Avg generation throughput: 844.7 tokens/s, Running: 32 reqs, Waiting: 19 reqs, GPU KV cache usage: 41.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:28:41 [loggers.py:236] Engine 000: Avg prompt throughput: 676.0 tokens/s, Avg generation throughput: 761.4 tokens/s, Running: 32 reqs, Waiting: 29 reqs, GPU KV cache usage: 38.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:28:51 [loggers.py:236] Engine 000: Avg prompt throughput: 626.5 tokens/s, Avg generation throughput: 812.6 tokens/s, Running: 32 reqs, Waiting: 41 reqs, GPU KV cache usage: 50.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:29:01 [loggers.py:236] Engine 000: Avg prompt throughput: 842.0 tokens/s, Avg generation throughput: 681.5 tokens/s, Running: 32 reqs, Waiting: 34 reqs, GPU KV cache usage: 36.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:29:11 [loggers.py:236] Engine 000: Avg prompt throughput: 790.7 tokens/s, Avg generation throughput: 748.6 tokens/s, Running: 32 reqs, Waiting: 42 reqs, GPU KV cache usage: 31.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:29:21 [loggers.py:236] Engine 000: Avg prompt throughput: 495.8 tokens/s, Avg generation throughput: 905.6 tokens/s, Running: 31 reqs, Waiting: 55 reqs, GPU KV cache usage: 34.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:29:31 [loggers.py:236] Engine 000: Avg prompt throughput: 713.1 tokens/s, Avg generation throughput: 790.4 tokens/s, Running: 32 reqs, Waiting: 55 reqs, GPU KV cache usage: 30.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:29:41 [loggers.py:236] Engine 000: Avg prompt throughput: 608.2 tokens/s, Avg generation throughput: 847.9 tokens/s, Running: 32 reqs, Waiting: 57 reqs, GPU KV cache usage: 34.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:29:51 [loggers.py:236] Engine 000: Avg prompt throughput: 874.2 tokens/s, Avg generation throughput: 687.9 tokens/s, Running: 32 reqs, Waiting: 49 reqs, GPU KV cache usage: 37.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:30:01 [loggers.py:236] Engine 000: Avg prompt throughput: 842.1 tokens/s, Avg generation throughput: 700.7 tokens/s, Running: 32 reqs, Waiting: 55 reqs, GPU KV cache usage: 35.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37636 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:30:11 [loggers.py:236] Engine 000: Avg prompt throughput: 754.0 tokens/s, Avg generation throughput: 764.7 tokens/s, Running: 31 reqs, Waiting: 61 reqs, GPU KV cache usage: 32.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:30:21 [loggers.py:236] Engine 000: Avg prompt throughput: 650.4 tokens/s, Avg generation throughput: 803.2 tokens/s, Running: 32 reqs, Waiting: 54 reqs, GPU KV cache usage: 31.5%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37686 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:30:31 [loggers.py:236] Engine 000: Avg prompt throughput: 886.2 tokens/s, Avg generation throughput: 684.7 tokens/s, Running: 32 reqs, Waiting: 49 reqs, GPU KV cache usage: 36.0%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47756 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:33000 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:30:41 [loggers.py:236] Engine 000: Avg prompt throughput: 661.1 tokens/s, Avg generation throughput: 767.9 tokens/s, Running: 31 reqs, Waiting: 55 reqs, GPU KV cache usage: 34.6%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:52184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50230 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:53682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:37558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:36878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:46438 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:32994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:47724 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:39528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:40250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50140 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44542 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:44552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:30:51 [loggers.py:236] Engine 000: Avg prompt throughput: 557.5 tokens/s, Avg generation throughput: 880.0 tokens/s, Running: 31 reqs, Waiting: 78 reqs, GPU KV cache usage: 36.0%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:31:01 [loggers.py:236] Engine 000: Avg prompt throughput: 944.9 tokens/s, Avg generation throughput: 614.4 tokens/s, Running: 31 reqs, Waiting: 40 reqs, GPU KV cache usage: 34.8%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:31:11 [loggers.py:236] Engine 000: Avg prompt throughput: 585.6 tokens/s, Avg generation throughput: 895.5 tokens/s, Running: 31 reqs, Waiting: 0 reqs, GPU KV cache usage: 30.5%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3017)[0;0m INFO 12-10 09:31:21 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 611.0 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.7%, Prefix cache hit rate: 0.1%
diff --git a/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..e84ded5
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 1446.3949257107452,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 0.6913741069083253,
+ "tokens_per_second": 512.5391321707765
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..e649879
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-db5fc148-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 184.17 \nTotal input tokens: 38358 \nTotal generated tokens: 40015 \nRequest throughput (req/s): 0.98 \nOutput token throughput (tok/s): 217.27 \nPeak output token throughput (tok/s): 541.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 425.55 \n---------------Time to First Token----------------\nMean TTFT (ms): 58.17 \nMedian TTFT (ms): 40.63 \nP99 TTFT (ms): 140.17 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 9.73 \nMedian TPOT (ms): 9.55 \nP99 TPOT (ms): 15.94 \n---------------Inter-token Latency----------------\nMean ITL (ms): 9.55 \nMedian ITL (ms): 9.45 \nP99 ITL (ms): 14.11 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..5e5dc50
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-c07c837c-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 189.59 \nTotal input tokens: 146694 \nTotal generated tokens: 153522 \nRequest throughput (req/s): 3.80 \nOutput token throughput (tok/s): 809.76 \nPeak output token throughput (tok/s): 1387.00 \nPeak concurrent requests: 40.00 \nTotal Token throughput (tok/s): 1583.51 \n---------------Time to First Token----------------\nMean TTFT (ms): 83.62 \nMedian TTFT (ms): 71.44 \nP99 TTFT (ms): 196.63 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 22.34 \nMedian TPOT (ms): 21.97 \nP99 TPOT (ms): 35.34 \n---------------Inter-token Latency----------------\nMean ITL (ms): 21.92 \nMedian ITL (ms): 19.31 \nP99 ITL (ms): 107.58 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
new file mode 100644
index 0000000..fc8e21f
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
@@ -0,0 +1,1026 @@
+WARNING 12-10 09:44:20 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:44:20 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:44:20 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'max_num_seqs': 64}
+[0;36m(APIServer pid=3923)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:44:29 [model.py:637] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:44:29 [model.py:1750] Using max model len 24576
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:44:29 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:37 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:38 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:49511 backend=nccl
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:38 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:38 [gpu_model_runner.py:3467] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:39 [compressed_tensors_wNa16.py:114] Using MarlinLinearKernel for CompressedTensorsWNA16
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:39 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:39 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:39 [compressed_tensors_moe.py:167] Using CompressedTensorsWNA16MarlinMoEMethod
+[0;36m(EngineCore_DP0 pid=3963)[0;0m WARNING 12-10 09:44:39 [compressed_tensors.py:721] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:01<00:03, 1.25s/it]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:07<00:08, 4.48s/it]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:13<00:04, 4.95s/it]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:18<00:00, 5.14s/it]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:18<00:00, 4.73s/it]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:44:59 [default_loader.py:308] Loading weights took 19.02 seconds
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:02 [gpu_model_runner.py:3549] Model loading took 15.6116 GiB memory and 22.703472 seconds
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:16 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/950831accb/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:16 [backends.py:715] Dynamo bytecode transform time: 13.68 s
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:17 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:33 [backends.py:288] Compiling a graph for dynamic shape takes 16.19 s
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:37 [monitor.py:34] torch.compile takes 29.88 s in total
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:38 [gpu_worker.py:359] Available KV cache memory: 5.08 GiB
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:38 [kv_cache_utils.py:1286] GPU KV cache size: 55,504 tokens
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:38 [kv_cache_utils.py:1291] Maximum concurrency for 24,576 tokens per request: 2.26x
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:01, 9.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:01, 9.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:01, 8.94it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:00<00:01, 9.11it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:01, 9.18it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:01, 9.23it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:00<00:01, 9.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:01, 9.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:00<00:01, 9.33it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:01<00:00, 9.34it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:01<00:00, 9.38it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:01<00:00, 9.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:01<00:00, 9.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:01<00:00, 9.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:01<00:00, 9.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:01<00:00, 9.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:01<00:00, 9.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:01<00:00, 9.50it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:02<00:00, 9.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:02<00:00, 9.33it/s]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 7.68it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:01, 8.47it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 8.78it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 8.93it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:00<00:00, 9.03it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 9.09it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:00<00:00, 9.14it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 9.18it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:01<00:00, 9.15it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:01<00:00, 9.17it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:01<00:00, 9.29it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:01<00:00, 9.06it/s]
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:42 [gpu_model_runner.py:4466] Graph capturing finished in 4 secs, took 0.32 GiB
+[0;36m(EngineCore_DP0 pid=3963)[0;0m INFO 12-10 09:45:42 [core.py:254] init engine (profile, create kv cache, warmup model) took 40.65 seconds
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:44 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=3923)[0;0m WARNING 12-10 09:45:44 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:44 [serving_responses.py:194] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:44 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [serving_completion.py:73] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:45:45 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=3923)[0;0m INFO: Started server process [3923]
+[0;36m(APIServer pid=3923)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=3923)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:48268 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:46:05 [loggers.py:236] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 11.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:46:15 [loggers.py:236] Engine 000: Avg prompt throughput: 115.5 tokens/s, Avg generation throughput: 208.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:46:25 [loggers.py:236] Engine 000: Avg prompt throughput: 207.6 tokens/s, Avg generation throughput: 185.4 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:46:35 [loggers.py:236] Engine 000: Avg prompt throughput: 193.4 tokens/s, Avg generation throughput: 169.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:46:45 [loggers.py:236] Engine 000: Avg prompt throughput: 309.4 tokens/s, Avg generation throughput: 153.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:46:55 [loggers.py:236] Engine 000: Avg prompt throughput: 205.4 tokens/s, Avg generation throughput: 216.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:47:05 [loggers.py:236] Engine 000: Avg prompt throughput: 438.1 tokens/s, Avg generation throughput: 182.5 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:36804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:47:15 [loggers.py:236] Engine 000: Avg prompt throughput: 257.7 tokens/s, Avg generation throughput: 253.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:47:25 [loggers.py:236] Engine 000: Avg prompt throughput: 369.4 tokens/s, Avg generation throughput: 207.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:47:35 [loggers.py:236] Engine 000: Avg prompt throughput: 174.9 tokens/s, Avg generation throughput: 381.0 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:46190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47700 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:47:45 [loggers.py:236] Engine 000: Avg prompt throughput: 332.3 tokens/s, Avg generation throughput: 256.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55740 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:47:55 [loggers.py:236] Engine 000: Avg prompt throughput: 45.4 tokens/s, Avg generation throughput: 243.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:48:05 [loggers.py:236] Engine 000: Avg prompt throughput: 201.2 tokens/s, Avg generation throughput: 163.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:48:15 [loggers.py:236] Engine 000: Avg prompt throughput: 242.5 tokens/s, Avg generation throughput: 279.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:42490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:48:25 [loggers.py:236] Engine 000: Avg prompt throughput: 103.3 tokens/s, Avg generation throughput: 257.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:48:35 [loggers.py:236] Engine 000: Avg prompt throughput: 93.6 tokens/s, Avg generation throughput: 57.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:48:45 [loggers.py:236] Engine 000: Avg prompt throughput: 106.2 tokens/s, Avg generation throughput: 227.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:48:55 [loggers.py:236] Engine 000: Avg prompt throughput: 140.1 tokens/s, Avg generation throughput: 182.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:39204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:39238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:39238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:39214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:49:05 [loggers.py:236] Engine 000: Avg prompt throughput: 299.7 tokens/s, Avg generation throughput: 275.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:49:15 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 100.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:49:25 [loggers.py:236] Engine 000: Avg prompt throughput: 84.0 tokens/s, Avg generation throughput: 62.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:49:35 [loggers.py:236] Engine 000: Avg prompt throughput: 904.2 tokens/s, Avg generation throughput: 551.0 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:49:45 [loggers.py:236] Engine 000: Avg prompt throughput: 1189.1 tokens/s, Avg generation throughput: 781.8 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:49:55 [loggers.py:236] Engine 000: Avg prompt throughput: 799.4 tokens/s, Avg generation throughput: 982.2 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:50:05 [loggers.py:236] Engine 000: Avg prompt throughput: 534.6 tokens/s, Avg generation throughput: 871.2 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:50:15 [loggers.py:236] Engine 000: Avg prompt throughput: 902.5 tokens/s, Avg generation throughput: 893.7 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:50:25 [loggers.py:236] Engine 000: Avg prompt throughput: 1066.5 tokens/s, Avg generation throughput: 896.8 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56484 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:50:35 [loggers.py:236] Engine 000: Avg prompt throughput: 615.8 tokens/s, Avg generation throughput: 847.9 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:50:45 [loggers.py:236] Engine 000: Avg prompt throughput: 820.1 tokens/s, Avg generation throughput: 860.9 tokens/s, Running: 31 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:50:55 [loggers.py:236] Engine 000: Avg prompt throughput: 702.4 tokens/s, Avg generation throughput: 1158.2 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47320 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:51:05 [loggers.py:236] Engine 000: Avg prompt throughput: 906.1 tokens/s, Avg generation throughput: 802.7 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:55070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:51:15 [loggers.py:236] Engine 000: Avg prompt throughput: 916.0 tokens/s, Avg generation throughput: 747.6 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:51:25 [loggers.py:236] Engine 000: Avg prompt throughput: 364.6 tokens/s, Avg generation throughput: 724.6 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:51:35 [loggers.py:236] Engine 000: Avg prompt throughput: 834.5 tokens/s, Avg generation throughput: 747.1 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.9%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:33102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:33102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:51:45 [loggers.py:236] Engine 000: Avg prompt throughput: 1052.8 tokens/s, Avg generation throughput: 747.8 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.5%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:33102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:51:55 [loggers.py:236] Engine 000: Avg prompt throughput: 551.0 tokens/s, Avg generation throughput: 680.2 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.5%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:33102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:33102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:52:05 [loggers.py:236] Engine 000: Avg prompt throughput: 780.0 tokens/s, Avg generation throughput: 811.9 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.1%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:52:15 [loggers.py:236] Engine 000: Avg prompt throughput: 846.9 tokens/s, Avg generation throughput: 791.3 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.7%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:33102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:60264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:60272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:60276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:56474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:33102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:47386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:59572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:38296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:60280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:60280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:49038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:60272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO: 127.0.0.1:54304 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=3923)[0;0m INFO 12-10 09:52:25 [loggers.py:236] Engine 000: Avg prompt throughput: 798.4 tokens/s, Avg generation throughput: 920.5 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.1%, Prefix cache hit rate: 0.1%
diff --git a/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
new file mode 100644
index 0000000..c31154d
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 306.69494965299964,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 3.2605688523121055,
+ "tokens_per_second": 2417.170549559942
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..73c88a4
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-bec359dc-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 191.18 \nTotal input tokens: 37841 \nTotal generated tokens: 38760 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 202.74 \nPeak output token throughput (tok/s): 418.00 \nPeak concurrent requests: 11.00 \nTotal Token throughput (tok/s): 400.67 \n---------------Time to First Token----------------\nMean TTFT (ms): 95.72 \nMedian TTFT (ms): 55.76 \nP99 TTFT (ms): 257.24 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 23.29 \nMedian TPOT (ms): 23.04 \nP99 TPOT (ms): 28.71 \n---------------Inter-token Latency----------------\nMean ITL (ms): 23.29 \nMedian ITL (ms): 22.01 \nP99 ITL (ms): 67.09 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..903dc61
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-acbcf111-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 199.30 \nTotal input tokens: 145810 \nTotal generated tokens: 151171 \nRequest throughput (req/s): 3.61 \nOutput token throughput (tok/s): 758.52 \nPeak output token throughput (tok/s): 1326.00 \nPeak concurrent requests: 52.00 \nTotal Token throughput (tok/s): 1490.14 \n---------------Time to First Token----------------\nMean TTFT (ms): 134.75 \nMedian TTFT (ms): 130.98 \nP99 TTFT (ms): 409.51 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 36.74 \nMedian TPOT (ms): 35.30 \nP99 TPOT (ms): 77.70 \n---------------Inter-token Latency----------------\nMean ITL (ms): 35.30 \nMedian ITL (ms): 26.57 \nP99 ITL (ms): 192.53 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
new file mode 100644
index 0000000..1a11313
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
@@ -0,0 +1,1023 @@
+WARNING 12-10 10:12:27 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:12:27 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:12:27 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 31800, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:12:36 [model.py:637] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:12:36 [model.py:1750] Using max model len 31800
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:12:36 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:12:46 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=31800, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:12:46 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:45209 backend=nccl
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:12:46 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:12:47 [gpu_model_runner.py:3467] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:12:47 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 4.68it/s]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.64it/s]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.32it/s]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.16s/it]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.05it/s]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:12:53 [default_loader.py:308] Loading weights took 3.89 seconds
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:12:53 [gpu_model_runner.py:3549] Model loading took 14.9889 GiB memory and 5.945220 seconds
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:00 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/d7ff4b2191/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:00 [backends.py:715] Dynamo bytecode transform time: 7.09 s
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:01 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:13 [backends.py:288] Compiling a graph for dynamic shape takes 11.36 s
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:14 [monitor.py:34] torch.compile takes 18.45 s in total
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:15 [gpu_worker.py:359] Available KV cache memory: 6.95 GiB
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:16 [kv_cache_utils.py:1286] GPU KV cache size: 56,880 tokens
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:16 [kv_cache_utils.py:1291] Maximum concurrency for 31,800 tokens per request: 1.79x
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:01, 15.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:00<00:00, 15.90it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 15.90it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 15.34it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:00<00:00, 16.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 16.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 16.77it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:00<00:00, 16.85it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:01<00:00, 17.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:01<00:00, 16.34it/s]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:00, 14.55it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 15.65it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 16.24it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 16.52it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:00<00:00, 16.66it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 16.43it/s]
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:18 [gpu_model_runner.py:4466] Graph capturing finished in 2 secs, took 0.21 GiB
+[0;36m(EngineCore_DP0 pid=4866)[0;0m INFO 12-10 10:13:18 [core.py:254] init engine (profile, create kv cache, warmup model) took 24.77 seconds
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:21 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=4826)[0;0m WARNING 12-10 10:13:21 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:21 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:21 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:22 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=4826)[0;0m INFO: Started server process [4826]
+[0;36m(APIServer pid=4826)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=4826)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:34868 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:13:52 [loggers.py:236] Engine 000: Avg prompt throughput: 84.5 tokens/s, Avg generation throughput: 88.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37792 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:14:02 [loggers.py:236] Engine 000: Avg prompt throughput: 132.8 tokens/s, Avg generation throughput: 144.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:14:12 [loggers.py:236] Engine 000: Avg prompt throughput: 143.0 tokens/s, Avg generation throughput: 202.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:14:22 [loggers.py:236] Engine 000: Avg prompt throughput: 285.5 tokens/s, Avg generation throughput: 155.0 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:44206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:14:32 [loggers.py:236] Engine 000: Avg prompt throughput: 215.0 tokens/s, Avg generation throughput: 177.6 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:44206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:14:42 [loggers.py:236] Engine 000: Avg prompt throughput: 333.9 tokens/s, Avg generation throughput: 246.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:44206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:44206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:44206 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:52786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:14:52 [loggers.py:236] Engine 000: Avg prompt throughput: 429.0 tokens/s, Avg generation throughput: 232.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:15:02 [loggers.py:236] Engine 000: Avg prompt throughput: 139.7 tokens/s, Avg generation throughput: 154.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:15:12 [loggers.py:236] Engine 000: Avg prompt throughput: 369.7 tokens/s, Avg generation throughput: 283.0 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:46838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:15:22 [loggers.py:236] Engine 000: Avg prompt throughput: 176.0 tokens/s, Avg generation throughput: 354.6 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:15:32 [loggers.py:236] Engine 000: Avg prompt throughput: 281.9 tokens/s, Avg generation throughput: 214.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:15:42 [loggers.py:236] Engine 000: Avg prompt throughput: 39.2 tokens/s, Avg generation throughput: 199.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:15:52 [loggers.py:236] Engine 000: Avg prompt throughput: 316.8 tokens/s, Avg generation throughput: 216.1 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:16:02 [loggers.py:236] Engine 000: Avg prompt throughput: 193.7 tokens/s, Avg generation throughput: 240.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:45260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:16:12 [loggers.py:236] Engine 000: Avg prompt throughput: 91.3 tokens/s, Avg generation throughput: 266.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:16:22 [loggers.py:236] Engine 000: Avg prompt throughput: 129.6 tokens/s, Avg generation throughput: 71.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:16:32 [loggers.py:236] Engine 000: Avg prompt throughput: 84.5 tokens/s, Avg generation throughput: 201.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:50820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:16:42 [loggers.py:236] Engine 000: Avg prompt throughput: 330.4 tokens/s, Avg generation throughput: 194.7 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:39022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:16:52 [loggers.py:236] Engine 000: Avg prompt throughput: 8.9 tokens/s, Avg generation throughput: 215.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:17:02 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 29.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:17:12 [loggers.py:236] Engine 000: Avg prompt throughput: 1.3 tokens/s, Avg generation throughput: 7.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:17:22 [loggers.py:236] Engine 000: Avg prompt throughput: 638.8 tokens/s, Avg generation throughput: 286.8 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:17:32 [loggers.py:236] Engine 000: Avg prompt throughput: 1071.7 tokens/s, Avg generation throughput: 677.5 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:17:42 [loggers.py:236] Engine 000: Avg prompt throughput: 916.4 tokens/s, Avg generation throughput: 842.8 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:17:52 [loggers.py:236] Engine 000: Avg prompt throughput: 697.0 tokens/s, Avg generation throughput: 926.7 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:18:02 [loggers.py:236] Engine 000: Avg prompt throughput: 790.5 tokens/s, Avg generation throughput: 817.0 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:18:12 [loggers.py:236] Engine 000: Avg prompt throughput: 944.0 tokens/s, Avg generation throughput: 881.8 tokens/s, Running: 43 reqs, Waiting: 0 reqs, GPU KV cache usage: 31.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:33388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:33388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:18:22 [loggers.py:236] Engine 000: Avg prompt throughput: 845.8 tokens/s, Avg generation throughput: 897.0 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:18:32 [loggers.py:236] Engine 000: Avg prompt throughput: 585.4 tokens/s, Avg generation throughput: 808.8 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58688 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:18:42 [loggers.py:236] Engine 000: Avg prompt throughput: 794.0 tokens/s, Avg generation throughput: 1052.7 tokens/s, Running: 45 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:33388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38022 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:18:52 [loggers.py:236] Engine 000: Avg prompt throughput: 647.4 tokens/s, Avg generation throughput: 943.8 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58684 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58732 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56252 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:33388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:19:02 [loggers.py:236] Engine 000: Avg prompt throughput: 867.8 tokens/s, Avg generation throughput: 781.6 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:33388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:19:12 [loggers.py:236] Engine 000: Avg prompt throughput: 820.2 tokens/s, Avg generation throughput: 640.5 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:19:22 [loggers.py:236] Engine 000: Avg prompt throughput: 724.5 tokens/s, Avg generation throughput: 634.5 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:19:32 [loggers.py:236] Engine 000: Avg prompt throughput: 784.8 tokens/s, Avg generation throughput: 901.0 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.1%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:19:42 [loggers.py:236] Engine 000: Avg prompt throughput: 829.9 tokens/s, Avg generation throughput: 666.5 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.6%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54306 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:19:52 [loggers.py:236] Engine 000: Avg prompt throughput: 542.4 tokens/s, Avg generation throughput: 761.5 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.4%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54318 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54380 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57606 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:20:02 [loggers.py:236] Engine 000: Avg prompt throughput: 684.2 tokens/s, Avg generation throughput: 770.1 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.9%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:48436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:38008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:41694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:58670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:57354 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:37994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:56242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:20:12 [loggers.py:236] Engine 000: Avg prompt throughput: 1282.3 tokens/s, Avg generation throughput: 769.9 tokens/s, Running: 37 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.0%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:55008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO: 127.0.0.1:54960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:20:22 [loggers.py:236] Engine 000: Avg prompt throughput: 112.4 tokens/s, Avg generation throughput: 840.4 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.6%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=4826)[0;0m INFO 12-10 10:20:32 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 218.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.1%
diff --git a/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
new file mode 100644
index 0000000..d1702fe
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 402.3294672546908,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 2.4855251265176648,
+ "tokens_per_second": 1830.1667164087519
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..dc8a33e
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-b23a48ff-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 184.20 \nTotal input tokens: 38756 \nTotal generated tokens: 39194 \nRequest throughput (req/s): 0.98 \nOutput token throughput (tok/s): 212.78 \nPeak output token throughput (tok/s): 495.00 \nPeak concurrent requests: 9.00 \nTotal Token throughput (tok/s): 423.18 \n---------------Time to First Token----------------\nMean TTFT (ms): 60.23 \nMedian TTFT (ms): 37.85 \nP99 TTFT (ms): 146.84 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 11.21 \nMedian TPOT (ms): 11.26 \nP99 TPOT (ms): 18.07 \n---------------Inter-token Latency----------------\nMean ITL (ms): 11.00 \nMedian ITL (ms): 10.95 \nP99 ITL (ms): 19.93 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..4ea3aa8
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-d2c7096a-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 190.29 \nTotal input tokens: 145540 \nTotal generated tokens: 151404 \nRequest throughput (req/s): 3.78 \nOutput token throughput (tok/s): 795.67 \nPeak output token throughput (tok/s): 1315.00 \nPeak concurrent requests: 36.00 \nTotal Token throughput (tok/s): 1560.51 \n---------------Time to First Token----------------\nMean TTFT (ms): 65.82 \nMedian TTFT (ms): 45.53 \nP99 TTFT (ms): 177.98 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 20.36 \nMedian TPOT (ms): 20.02 \nP99 TPOT (ms): 31.44 \n---------------Inter-token Latency----------------\nMean ITL (ms): 20.10 \nMedian ITL (ms): 18.54 \nP99 ITL (ms): 90.23 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_server.log b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_server.log
new file mode 100644
index 0000000..7036a6f
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_server.log
@@ -0,0 +1,1022 @@
+WARNING 12-10 08:45:12 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:45:12 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:45:12 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 16384, 'max_num_seqs': 32}
+[0;36m(APIServer pid=1916)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:45:19 [model.py:637] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=1916)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 3.73it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 11.17it/s]
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:45:20 [model.py:1750] Using max model len 16384
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:45:20 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:45:20 [config.py:274] Overriding max cuda graph capture size to 1024 for performance.
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:29 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=16384, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:30 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:59799 backend=nccl
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:30 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:30 [gpu_model_runner.py:3467] Starting to load model openai/gpt-oss-20b...
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:31 [cuda.py:411] Using TRITON_ATTN attention backend out of potential backends: ['TRITON_ATTN']
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:31 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:31 [mxfp4.py:162] Using Marlin backend
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:01, 1.40it/s]
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:01<00:00, 1.23it/s]
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.18it/s]
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.21it/s]
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:34 [default_loader.py:308] Loading weights took 2.57 seconds
+[0;36m(EngineCore_DP0 pid=1960)[0;0m WARNING 12-10 08:45:34 [marlin_utils_fp4.py:226] Your GPU does not have native support for FP4 computation but FP4 quantization is being used. Weight-only FP4 compression will be used leveraging the Marlin kernel. This may degrade performance for compute-heavy workloads.
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:35 [gpu_model_runner.py:3549] Model loading took 13.7194 GiB memory and 4.366336 seconds
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:41 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/9e0957b12c/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:41 [backends.py:715] Dynamo bytecode transform time: 5.94 s
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:42 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:50 [backends.py:288] Compiling a graph for dynamic shape takes 7.71 s
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:54 [monitor.py:34] torch.compile takes 13.66 s in total
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:55 [gpu_worker.py:359] Available KV cache memory: 7.12 GiB
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:55 [kv_cache_utils.py:1286] GPU KV cache size: 155,584 tokens
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:45:55 [kv_cache_utils.py:1291] Maximum concurrency for 16,384 tokens per request: 16.75x
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 1%| | 1/83 [00:00<00:12, 6.72it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:11, 6.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 4%|▎ | 3/83 [00:00<00:11, 6.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:00<00:11, 6.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 6%|▌ | 5/83 [00:00<00:11, 6.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:00<00:10, 7.02it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 8%|▊ | 7/83 [00:01<00:10, 7.05it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:01<00:10, 7.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 9/83 [00:01<00:10, 7.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:01<00:09, 7.33it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 13%|█▎ | 11/83 [00:01<00:09, 7.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:01<00:09, 7.48it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 13/83 [00:01<00:09, 7.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:01<00:09, 7.65it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 18%|█▊ | 15/83 [00:02<00:08, 7.74it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:02<00:08, 7.77it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 20%|██ | 17/83 [00:02<00:08, 7.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:02<00:08, 8.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 23%|██▎ | 19/83 [00:02<00:07, 8.13it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:02<00:07, 8.16it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 25%|██▌ | 21/83 [00:02<00:07, 8.25it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:02<00:07, 8.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 28%|██▊ | 23/83 [00:03<00:07, 8.40it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:03<00:06, 8.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 30%|███ | 25/83 [00:03<00:06, 8.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:03<00:06, 8.85it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 33%|███▎ | 27/83 [00:03<00:06, 9.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:03<00:05, 9.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 35%|███▍ | 29/83 [00:03<00:05, 9.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▌ | 30/83 [00:03<00:05, 9.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 31/83 [00:03<00:05, 9.58it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 39%|███▊ | 32/83 [00:03<00:05, 9.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 41%|████ | 34/83 [00:04<00:04, 10.20it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 43%|████▎ | 36/83 [00:04<00:04, 10.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 46%|████▌ | 38/83 [00:04<00:04, 10.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 48%|████▊ | 40/83 [00:04<00:03, 11.04it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 51%|█████ | 42/83 [00:04<00:03, 11.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 44/83 [00:04<00:03, 11.94it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▌ | 46/83 [00:05<00:02, 12.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 48/83 [00:05<00:02, 13.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 60%|██████ | 50/83 [00:05<00:02, 13.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 52/83 [00:05<00:02, 14.05it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 65%|██████▌ | 54/83 [00:05<00:01, 14.55it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 67%|██████▋ | 56/83 [00:05<00:01, 15.01it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 70%|██████▉ | 58/83 [00:05<00:01, 15.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 72%|███████▏ | 60/83 [00:06<00:01, 15.87it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 75%|███████▍ | 62/83 [00:06<00:01, 16.11it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 77%|███████▋ | 64/83 [00:06<00:01, 16.25it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 80%|███████▉ | 66/83 [00:06<00:01, 16.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:06<00:00, 16.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 70/83 [00:06<00:00, 16.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 87%|████████▋ | 72/83 [00:06<00:00, 16.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:06<00:00, 16.89it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 92%|█████████▏| 76/83 [00:06<00:00, 17.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 94%|█████████▍| 78/83 [00:07<00:00, 17.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 96%|█████████▋| 80/83 [00:07<00:00, 17.15it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 99%|█████████▉| 82/83 [00:07<00:00, 16.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:07<00:00, 11.23it/s]
+[0;36m(EngineCore_DP0 pid=1960)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 14%|█▍ | 1/7 [00:01<00:09, 1.66s/it]
Capturing CUDA graphs (decode, FULL): 29%|██▊ | 2/7 [00:02<00:06, 1.23s/it]
Capturing CUDA graphs (decode, FULL): 57%|█████▋ | 4/7 [00:04<00:03, 1.06s/it]
Capturing CUDA graphs (decode, FULL): 86%|████████▌ | 6/7 [00:04<00:00, 1.68it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:06<00:00, 1.22it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:06<00:00, 1.14it/s]
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:46:09 [gpu_model_runner.py:4466] Graph capturing finished in 14 secs, took 0.63 GiB
+[0;36m(EngineCore_DP0 pid=1960)[0;0m INFO 12-10 08:46:09 [core.py:254] init engine (profile, create kv cache, warmup model) took 33.89 seconds
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:11 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=1916)[0;0m WARNING 12-10 08:46:11 [serving_responses.py:215] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:14 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=1916)[0;0m INFO: Started server process [1916]
+[0;36m(APIServer pid=1916)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=1916)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:38426 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:45 [loggers.py:236] Engine 000: Avg prompt throughput: 83.7 tokens/s, Avg generation throughput: 161.4 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:46:55 [loggers.py:236] Engine 000: Avg prompt throughput: 132.9 tokens/s, Avg generation throughput: 170.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45178 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:47:05 [loggers.py:236] Engine 000: Avg prompt throughput: 214.2 tokens/s, Avg generation throughput: 178.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:47:15 [loggers.py:236] Engine 000: Avg prompt throughput: 290.4 tokens/s, Avg generation throughput: 150.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:47:25 [loggers.py:236] Engine 000: Avg prompt throughput: 269.8 tokens/s, Avg generation throughput: 203.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:47:35 [loggers.py:236] Engine 000: Avg prompt throughput: 240.4 tokens/s, Avg generation throughput: 190.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:60260 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:37486 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:47:45 [loggers.py:236] Engine 000: Avg prompt throughput: 476.2 tokens/s, Avg generation throughput: 317.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:47:55 [loggers.py:236] Engine 000: Avg prompt throughput: 261.0 tokens/s, Avg generation throughput: 80.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:48:05 [loggers.py:236] Engine 000: Avg prompt throughput: 183.3 tokens/s, Avg generation throughput: 405.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58204 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58228 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:48:15 [loggers.py:236] Engine 000: Avg prompt throughput: 406.7 tokens/s, Avg generation throughput: 341.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46558 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:48:25 [loggers.py:236] Engine 000: Avg prompt throughput: 128.6 tokens/s, Avg generation throughput: 145.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:48:35 [loggers.py:236] Engine 000: Avg prompt throughput: 108.9 tokens/s, Avg generation throughput: 202.3 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:36434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:36434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:48:45 [loggers.py:236] Engine 000: Avg prompt throughput: 245.0 tokens/s, Avg generation throughput: 197.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:36434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:36434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:36434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:36434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:48:55 [loggers.py:236] Engine 000: Avg prompt throughput: 170.6 tokens/s, Avg generation throughput: 359.9 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58240 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:36434 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:49:05 [loggers.py:236] Engine 000: Avg prompt throughput: 104.4 tokens/s, Avg generation throughput: 94.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50984 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:49:15 [loggers.py:236] Engine 000: Avg prompt throughput: 109.5 tokens/s, Avg generation throughput: 97.9 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57746 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:49:25 [loggers.py:236] Engine 000: Avg prompt throughput: 131.5 tokens/s, Avg generation throughput: 265.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:49:35 [loggers.py:236] Engine 000: Avg prompt throughput: 244.3 tokens/s, Avg generation throughput: 227.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:50982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:49:45 [loggers.py:236] Engine 000: Avg prompt throughput: 75.3 tokens/s, Avg generation throughput: 141.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:49:55 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:50:05 [loggers.py:236] Engine 000: Avg prompt throughput: 637.9 tokens/s, Avg generation throughput: 452.2 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 13.5%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:50:15 [loggers.py:236] Engine 000: Avg prompt throughput: 1216.5 tokens/s, Avg generation throughput: 790.4 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 31.2%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:50:25 [loggers.py:236] Engine 000: Avg prompt throughput: 863.1 tokens/s, Avg generation throughput: 948.4 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 39.6%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:50:35 [loggers.py:236] Engine 000: Avg prompt throughput: 666.2 tokens/s, Avg generation throughput: 849.1 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 44.8%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:50:45 [loggers.py:236] Engine 000: Avg prompt throughput: 814.9 tokens/s, Avg generation throughput: 790.6 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 46.1%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:50:55 [loggers.py:236] Engine 000: Avg prompt throughput: 962.0 tokens/s, Avg generation throughput: 944.4 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 41.2%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:51:05 [loggers.py:236] Engine 000: Avg prompt throughput: 821.8 tokens/s, Avg generation throughput: 817.4 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 37.7%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:51:15 [loggers.py:236] Engine 000: Avg prompt throughput: 576.8 tokens/s, Avg generation throughput: 793.2 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 35.6%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:51:25 [loggers.py:236] Engine 000: Avg prompt throughput: 828.7 tokens/s, Avg generation throughput: 1155.7 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 33.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:51:35 [loggers.py:236] Engine 000: Avg prompt throughput: 650.3 tokens/s, Avg generation throughput: 873.2 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 31.2%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40874 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57440 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:45194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57436 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52896 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:51:45 [loggers.py:236] Engine 000: Avg prompt throughput: 1187.4 tokens/s, Avg generation throughput: 715.4 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 29.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:51:55 [loggers.py:236] Engine 000: Avg prompt throughput: 418.6 tokens/s, Avg generation throughput: 704.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 28.1%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:52:05 [loggers.py:236] Engine 000: Avg prompt throughput: 801.0 tokens/s, Avg generation throughput: 703.3 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 26.6%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:52:15 [loggers.py:236] Engine 000: Avg prompt throughput: 953.1 tokens/s, Avg generation throughput: 801.5 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 25.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:52:25 [loggers.py:236] Engine 000: Avg prompt throughput: 702.7 tokens/s, Avg generation throughput: 656.4 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 23.9%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:52:35 [loggers.py:236] Engine 000: Avg prompt throughput: 677.7 tokens/s, Avg generation throughput: 768.8 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 22.9%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:52:45 [loggers.py:236] Engine 000: Avg prompt throughput: 656.2 tokens/s, Avg generation throughput: 813.9 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 22.0%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:58938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47084 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46936 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:40870 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:53704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:52:55 [loggers.py:236] Engine 000: Avg prompt throughput: 1049.0 tokens/s, Avg generation throughput: 769.4 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 20.8%
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:47128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52920 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:46960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO: 127.0.0.1:52930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=1916)[0;0m INFO 12-10 08:53:05 [loggers.py:236] Engine 000: Avg prompt throughput: 70.1 tokens/s, Avg generation throughput: 782.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 20.7%
diff --git a/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_throughput.json
new file mode 100644
index 0000000..38848a6
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-3090/openai_gpt-oss-20b_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 357.56243500020355,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 2.7967143696161223,
+ "tokens_per_second": 2066.190202557434
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..42c7ea6
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-96888f07-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 190.82 \nTotal input tokens: 38358 \nTotal generated tokens: 40296 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 211.17 \nPeak output token throughput (tok/s): 443.00 \nPeak concurrent requests: 11.00 \nTotal Token throughput (tok/s): 412.18 \n---------------Time to First Token----------------\nMean TTFT (ms): 64.41 \nMedian TTFT (ms): 50.82 \nP99 TTFT (ms): 144.55 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 21.14 \nMedian TPOT (ms): 20.98 \nP99 TPOT (ms): 23.59 \n---------------Inter-token Latency----------------\nMean ITL (ms): 21.12 \nMedian ITL (ms): 20.65 \nP99 ITL (ms): 29.23 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..d623116
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-bd32e896-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 194.55 \nTotal input tokens: 146694 \nTotal generated tokens: 155576 \nRequest throughput (req/s): 3.70 \nOutput token throughput (tok/s): 799.66 \nPeak output token throughput (tok/s): 1300.00 \nPeak concurrent requests: 43.00 \nTotal Token throughput (tok/s): 1553.66 \n---------------Time to First Token----------------\nMean TTFT (ms): 92.77 \nMedian TTFT (ms): 68.07 \nP99 TTFT (ms): 484.98 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 25.78 \nMedian TPOT (ms): 25.52 \nP99 TPOT (ms): 34.37 \n---------------Inter-token Latency----------------\nMean ITL (ms): 25.65 \nMedian ITL (ms): 23.39 \nP99 ITL (ms): 86.03 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..49fed97
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1024 @@
+WARNING 12-09 21:03:27 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:03:27 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:03:27 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 4096, 'gpu_memory_utilization': 0.86, 'max_num_seqs': 32}
+[0;36m(APIServer pid=31169)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:03:34 [model.py:637] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:03:34 [model.py:1750] Using max model len 4096
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:03:34 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:41 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=4096, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 64, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:42 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:34731 backend=nccl
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:42 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:42 [gpu_model_runner.py:3467] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:43 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 4.87it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:00<00:00, 2.32it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:01<00:00, 1.78it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.59it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.78it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:46 [default_loader.py:308] Loading weights took 2.41 seconds
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:46 [gpu_model_runner.py:3549] Model loading took 15.3388 GiB memory and 3.593561 seconds
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:59 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/57323a05ea/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:03:59 [backends.py:715] Dynamo bytecode transform time: 12.33 s
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:00 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:09 [backends.py:288] Compiling a graph for dynamic shape takes 8.80 s
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:15 [monitor.py:34] torch.compile takes 21.13 s in total
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:16 [gpu_worker.py:359] Available KV cache memory: 4.43 GiB
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:16 [kv_cache_utils.py:1286] GPU KV cache size: 29,008 tokens
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:16 [kv_cache_utils.py:1291] Maximum concurrency for 4,096 tokens per request: 7.08x
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 3/11 [00:00<00:00, 21.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▍ | 6/11 [00:00<00:00, 22.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 9/11 [00:00<00:00, 22.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 11/11 [00:00<00:00, 22.34it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 29%|██▊ | 2/7 [00:00<00:00, 19.34it/s]
Capturing CUDA graphs (decode, FULL): 71%|███████▏ | 5/7 [00:00<00:00, 23.41it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:00<00:00, 23.77it/s]
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:18 [gpu_model_runner.py:4466] Graph capturing finished in 1 secs, took 0.16 GiB
+[0;36m(EngineCore_DP0 pid=31480)[0;0m INFO 12-09 21:04:18 [core.py:254] init engine (profile, create kv cache, warmup model) took 31.35 seconds
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=31169)[0;0m WARNING 12-09 21:04:20 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:20 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=31169)[0;0m INFO: Started server process [31169]
+[0;36m(APIServer pid=31169)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=31169)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40408 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:41 [loggers.py:236] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 8.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:04:51 [loggers.py:236] Engine 000: Avg prompt throughput: 115.5 tokens/s, Avg generation throughput: 124.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:52018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:05:01 [loggers.py:236] Engine 000: Avg prompt throughput: 206.3 tokens/s, Avg generation throughput: 190.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:05:11 [loggers.py:236] Engine 000: Avg prompt throughput: 118.5 tokens/s, Avg generation throughput: 163.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:05:21 [loggers.py:236] Engine 000: Avg prompt throughput: 323.1 tokens/s, Avg generation throughput: 183.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:05:31 [loggers.py:236] Engine 000: Avg prompt throughput: 253.3 tokens/s, Avg generation throughput: 210.5 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:05:41 [loggers.py:236] Engine 000: Avg prompt throughput: 386.6 tokens/s, Avg generation throughput: 210.6 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42066 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:05:51 [loggers.py:236] Engine 000: Avg prompt throughput: 323.8 tokens/s, Avg generation throughput: 268.6 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:06:01 [loggers.py:236] Engine 000: Avg prompt throughput: 343.8 tokens/s, Avg generation throughput: 145.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:06:11 [loggers.py:236] Engine 000: Avg prompt throughput: 172.3 tokens/s, Avg generation throughput: 359.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:06:21 [loggers.py:236] Engine 000: Avg prompt throughput: 358.8 tokens/s, Avg generation throughput: 338.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:06:31 [loggers.py:236] Engine 000: Avg prompt throughput: 34.6 tokens/s, Avg generation throughput: 187.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:06:41 [loggers.py:236] Engine 000: Avg prompt throughput: 213.8 tokens/s, Avg generation throughput: 212.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:40844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:06:51 [loggers.py:236] Engine 000: Avg prompt throughput: 234.9 tokens/s, Avg generation throughput: 213.0 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:45454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:07:01 [loggers.py:236] Engine 000: Avg prompt throughput: 110.0 tokens/s, Avg generation throughput: 318.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:39972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:07:11 [loggers.py:236] Engine 000: Avg prompt throughput: 94.5 tokens/s, Avg generation throughput: 119.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34912 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:07:21 [loggers.py:236] Engine 000: Avg prompt throughput: 81.6 tokens/s, Avg generation throughput: 118.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:07:31 [loggers.py:236] Engine 000: Avg prompt throughput: 164.7 tokens/s, Avg generation throughput: 243.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:44944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:44946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33168 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:07:41 [loggers.py:236] Engine 000: Avg prompt throughput: 251.1 tokens/s, Avg generation throughput: 222.4 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:33186 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:07:51 [loggers.py:236] Engine 000: Avg prompt throughput: 48.6 tokens/s, Avg generation throughput: 190.7 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:08:01 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 10.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:08:11 [loggers.py:236] Engine 000: Avg prompt throughput: 220.0 tokens/s, Avg generation throughput: 98.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:08:21 [loggers.py:236] Engine 000: Avg prompt throughput: 1019.5 tokens/s, Avg generation throughput: 588.5 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 23.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:08:31 [loggers.py:236] Engine 000: Avg prompt throughput: 1216.1 tokens/s, Avg generation throughput: 853.1 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 40.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34172 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:08:41 [loggers.py:236] Engine 000: Avg prompt throughput: 648.4 tokens/s, Avg generation throughput: 995.1 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 38.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:08:51 [loggers.py:236] Engine 000: Avg prompt throughput: 534.4 tokens/s, Avg generation throughput: 819.4 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 22.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:09:01 [loggers.py:236] Engine 000: Avg prompt throughput: 962.2 tokens/s, Avg generation throughput: 959.0 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 37.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:09:11 [loggers.py:236] Engine 000: Avg prompt throughput: 1031.9 tokens/s, Avg generation throughput: 893.3 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 24.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:09:21 [loggers.py:236] Engine 000: Avg prompt throughput: 570.4 tokens/s, Avg generation throughput: 799.3 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:09:31 [loggers.py:236] Engine 000: Avg prompt throughput: 869.6 tokens/s, Avg generation throughput: 991.5 tokens/s, Running: 32 reqs, Waiting: 3 reqs, GPU KV cache usage: 32.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:09:41 [loggers.py:236] Engine 000: Avg prompt throughput: 636.1 tokens/s, Avg generation throughput: 1121.4 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 33.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:09:51 [loggers.py:236] Engine 000: Avg prompt throughput: 859.3 tokens/s, Avg generation throughput: 786.8 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:10:01 [loggers.py:236] Engine 000: Avg prompt throughput: 948.0 tokens/s, Avg generation throughput: 688.8 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 21.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:10:11 [loggers.py:236] Engine 000: Avg prompt throughput: 574.2 tokens/s, Avg generation throughput: 701.7 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34194 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:10:21 [loggers.py:236] Engine 000: Avg prompt throughput: 854.5 tokens/s, Avg generation throughput: 852.7 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.4%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:10:31 [loggers.py:236] Engine 000: Avg prompt throughput: 846.6 tokens/s, Avg generation throughput: 765.6 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.1%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:10:41 [loggers.py:236] Engine 000: Avg prompt throughput: 613.5 tokens/s, Avg generation throughput: 696.1 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 20.5%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:10:51 [loggers.py:236] Engine 000: Avg prompt throughput: 730.2 tokens/s, Avg generation throughput: 825.2 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 32.3%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34100 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58752 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34196 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:11:01 [loggers.py:236] Engine 000: Avg prompt throughput: 1098.9 tokens/s, Avg generation throughput: 786.1 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 26.8%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34832 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58736 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:43126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:60810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34314 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:58828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:42556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34340 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34146 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:47994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:47996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:48012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:34330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO: 127.0.0.1:59488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:11:11 [loggers.py:236] Engine 000: Avg prompt throughput: 434.8 tokens/s, Avg generation throughput: 979.0 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 28.6%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=31169)[0;0m INFO 12-09 21:11:21 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 362.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.1%
diff --git a/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..093bec2
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 889.9994421191514,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 1.1235962099245025,
+ "tokens_per_second": 832.9600726881711
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..8ca7e54
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-4cdb5c8d-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 183.35 \nTotal input tokens: 38358 \nTotal generated tokens: 39718 \nRequest throughput (req/s): 0.98 \nOutput token throughput (tok/s): 216.63 \nPeak output token throughput (tok/s): 666.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 425.84 \n---------------Time to First Token----------------\nMean TTFT (ms): 37.48 \nMedian TTFT (ms): 30.33 \nP99 TTFT (ms): 77.17 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 7.36 \nMedian TPOT (ms): 7.11 \nP99 TPOT (ms): 11.45 \n---------------Inter-token Latency----------------\nMean ITL (ms): 7.26 \nMedian ITL (ms): 6.53 \nP99 ITL (ms): 11.05 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..0211464
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-81c0edc1-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 186.20 \nTotal input tokens: 146694 \nTotal generated tokens: 153033 \nRequest throughput (req/s): 3.87 \nOutput token throughput (tok/s): 821.89 \nPeak output token throughput (tok/s): 1471.00 \nPeak concurrent requests: 30.00 \nTotal Token throughput (tok/s): 1609.74 \n---------------Time to First Token----------------\nMean TTFT (ms): 44.36 \nMedian TTFT (ms): 42.14 \nP99 TTFT (ms): 82.58 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 13.89 \nMedian TPOT (ms): 13.91 \nP99 TPOT (ms): 17.75 \n---------------Inter-token Latency----------------\nMean ITL (ms): 13.74 \nMedian ITL (ms): 13.29 \nP99 ITL (ms): 38.23 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
new file mode 100644
index 0000000..29c36f8
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
@@ -0,0 +1,1026 @@
+WARNING 12-09 21:18:19 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:18:19 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:18:19 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'max_num_seqs': 64}
+[0;36m(APIServer pid=35981)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:18:25 [model.py:637] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:18:25 [model.py:1750] Using max model len 24576
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:18:26 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:33 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:34 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:58697 backend=nccl
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:34 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:34 [gpu_model_runner.py:3467] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:35 [compressed_tensors_wNa16.py:114] Using MarlinLinearKernel for CompressedTensorsWNA16
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:35 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:35 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:35 [compressed_tensors_moe.py:167] Using CompressedTensorsWNA16MarlinMoEMethod
+[0;36m(EngineCore_DP0 pid=36287)[0;0m WARNING 12-09 21:18:35 [compressed_tensors.py:721] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:02, 1.03it/s]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:05<00:05, 2.81s/it]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:09<00:03, 3.72s/it]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:14<00:00, 4.16s/it]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:14<00:00, 3.68s/it]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:51 [default_loader.py:308] Loading weights took 14.87 seconds
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:18:53 [gpu_model_runner.py:3549] Model loading took 15.6116 GiB memory and 17.565601 seconds
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:02 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/731dc1b1d0/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:02 [backends.py:715] Dynamo bytecode transform time: 9.12 s
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:03 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:15 [backends.py:288] Compiling a graph for dynamic shape takes 11.61 s
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:17 [monitor.py:34] torch.compile takes 20.73 s in total
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:18 [gpu_worker.py:359] Available KV cache memory: 5.07 GiB
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:18 [kv_cache_utils.py:1286] GPU KV cache size: 55,360 tokens
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:18 [kv_cache_utils.py:1291] Maximum concurrency for 24,576 tokens per request: 2.25x
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:01, 12.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:00<00:01, 13.25it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 13.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 14.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:00<00:00, 14.68it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 14.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 15.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:01<00:00, 15.12it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:01<00:00, 15.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:01<00:00, 14.65it/s]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:00, 13.27it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 14.49it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 14.67it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 14.73it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:00<00:00, 15.07it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 14.90it/s]
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:21 [gpu_model_runner.py:4466] Graph capturing finished in 3 secs, took 0.39 GiB
+[0;36m(EngineCore_DP0 pid=36287)[0;0m INFO 12-09 21:19:21 [core.py:254] init engine (profile, create kv cache, warmup model) took 28.22 seconds
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=35981)[0;0m WARNING 12-09 21:19:23 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [serving_responses.py:194] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [serving_completion.py:73] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:23 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=35981)[0;0m INFO: Started server process [35981]
+[0;36m(APIServer pid=35981)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=35981)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39034 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:44 [loggers.py:236] Engine 000: Avg prompt throughput: 4.8 tokens/s, Avg generation throughput: 31.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41180 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:19:54 [loggers.py:236] Engine 000: Avg prompt throughput: 133.8 tokens/s, Avg generation throughput: 218.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:43864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:20:04 [loggers.py:236] Engine 000: Avg prompt throughput: 224.6 tokens/s, Avg generation throughput: 193.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:20:14 [loggers.py:236] Engine 000: Avg prompt throughput: 279.4 tokens/s, Avg generation throughput: 156.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:20:24 [loggers.py:236] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 214.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:20:34 [loggers.py:236] Engine 000: Avg prompt throughput: 374.5 tokens/s, Avg generation throughput: 205.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:20:44 [loggers.py:236] Engine 000: Avg prompt throughput: 343.4 tokens/s, Avg generation throughput: 233.6 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:47310 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:20:54 [loggers.py:236] Engine 000: Avg prompt throughput: 278.0 tokens/s, Avg generation throughput: 169.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:21:04 [loggers.py:236] Engine 000: Avg prompt throughput: 295.0 tokens/s, Avg generation throughput: 308.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:21:14 [loggers.py:236] Engine 000: Avg prompt throughput: 176.4 tokens/s, Avg generation throughput: 382.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:21:24 [loggers.py:236] Engine 000: Avg prompt throughput: 284.7 tokens/s, Avg generation throughput: 176.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:21:34 [loggers.py:236] Engine 000: Avg prompt throughput: 40.2 tokens/s, Avg generation throughput: 207.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:21:44 [loggers.py:236] Engine 000: Avg prompt throughput: 229.6 tokens/s, Avg generation throughput: 168.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:44794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:44794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:21:54 [loggers.py:236] Engine 000: Avg prompt throughput: 219.1 tokens/s, Avg generation throughput: 344.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:44794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:44794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53902 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:22:04 [loggers.py:236] Engine 000: Avg prompt throughput: 156.4 tokens/s, Avg generation throughput: 179.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:53876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:44794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:22:14 [loggers.py:236] Engine 000: Avg prompt throughput: 70.6 tokens/s, Avg generation throughput: 18.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50828 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:22:24 [loggers.py:236] Engine 000: Avg prompt throughput: 172.9 tokens/s, Avg generation throughput: 301.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:22:34 [loggers.py:236] Engine 000: Avg prompt throughput: 73.3 tokens/s, Avg generation throughput: 134.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33884 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:50834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:45954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33878 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:22:44 [loggers.py:236] Engine 000: Avg prompt throughput: 263.0 tokens/s, Avg generation throughput: 316.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:22:54 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 21.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:23:04 [loggers.py:236] Engine 000: Avg prompt throughput: 428.2 tokens/s, Avg generation throughput: 306.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:23:14 [loggers.py:236] Engine 000: Avg prompt throughput: 1212.1 tokens/s, Avg generation throughput: 758.4 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:23:24 [loggers.py:236] Engine 000: Avg prompt throughput: 971.3 tokens/s, Avg generation throughput: 946.7 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:23:34 [loggers.py:236] Engine 000: Avg prompt throughput: 645.3 tokens/s, Avg generation throughput: 960.7 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:23:44 [loggers.py:236] Engine 000: Avg prompt throughput: 667.4 tokens/s, Avg generation throughput: 757.3 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:23:54 [loggers.py:236] Engine 000: Avg prompt throughput: 931.7 tokens/s, Avg generation throughput: 987.9 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:24:04 [loggers.py:236] Engine 000: Avg prompt throughput: 983.9 tokens/s, Avg generation throughput: 846.4 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:24:14 [loggers.py:236] Engine 000: Avg prompt throughput: 480.2 tokens/s, Avg generation throughput: 747.9 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:24:24 [loggers.py:236] Engine 000: Avg prompt throughput: 974.8 tokens/s, Avg generation throughput: 1112.9 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:24:34 [loggers.py:236] Engine 000: Avg prompt throughput: 544.8 tokens/s, Avg generation throughput: 988.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:24:44 [loggers.py:236] Engine 000: Avg prompt throughput: 950.1 tokens/s, Avg generation throughput: 724.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:24:54 [loggers.py:236] Engine 000: Avg prompt throughput: 833.6 tokens/s, Avg generation throughput: 646.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34594 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:25:04 [loggers.py:236] Engine 000: Avg prompt throughput: 575.4 tokens/s, Avg generation throughput: 711.7 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:25:14 [loggers.py:236] Engine 000: Avg prompt throughput: 839.8 tokens/s, Avg generation throughput: 869.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:25:24 [loggers.py:236] Engine 000: Avg prompt throughput: 870.2 tokens/s, Avg generation throughput: 614.8 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:25:34 [loggers.py:236] Engine 000: Avg prompt throughput: 832.2 tokens/s, Avg generation throughput: 776.0 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.2%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:25:44 [loggers.py:236] Engine 000: Avg prompt throughput: 466.5 tokens/s, Avg generation throughput: 854.6 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.9%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:25:54 [loggers.py:236] Engine 000: Avg prompt throughput: 1159.2 tokens/s, Avg generation throughput: 714.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.0%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39604 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34648 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34664 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:48718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:42586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:33514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:48720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:48720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:34602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:41248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO: 127.0.0.1:39578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=35981)[0;0m INFO 12-09 21:26:04 [loggers.py:236] Engine 000: Avg prompt throughput: 302.7 tokens/s, Avg generation throughput: 987.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.1%
diff --git a/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
new file mode 100644
index 0000000..8cb379e
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 215.380199868232,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 4.642952326220295,
+ "tokens_per_second": 3441.9784198061966
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..74f3ee8
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-04789567-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 188.87 \nTotal input tokens: 37841 \nTotal generated tokens: 38900 \nRequest throughput (req/s): 0.95 \nOutput token throughput (tok/s): 205.96 \nPeak output token throughput (tok/s): 465.00 \nPeak concurrent requests: 10.00 \nTotal Token throughput (tok/s): 406.32 \n---------------Time to First Token----------------\nMean TTFT (ms): 52.16 \nMedian TTFT (ms): 44.25 \nP99 TTFT (ms): 112.79 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 17.68 \nMedian TPOT (ms): 17.62 \nP99 TPOT (ms): 19.35 \n---------------Inter-token Latency----------------\nMean ITL (ms): 17.64 \nMedian ITL (ms): 17.35 \nP99 ITL (ms): 23.66 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..a8b55d3
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-b098e8e8-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 193.10 \nTotal input tokens: 145810 \nTotal generated tokens: 152091 \nRequest throughput (req/s): 3.73 \nOutput token throughput (tok/s): 787.62 \nPeak output token throughput (tok/s): 1351.00 \nPeak concurrent requests: 39.00 \nTotal Token throughput (tok/s): 1542.71 \n---------------Time to First Token----------------\nMean TTFT (ms): 57.32 \nMedian TTFT (ms): 49.20 \nP99 TTFT (ms): 126.64 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 20.27 \nMedian TPOT (ms): 20.07 \nP99 TPOT (ms): 27.95 \n---------------Inter-token Latency----------------\nMean ITL (ms): 20.15 \nMedian ITL (ms): 19.30 \nP99 ITL (ms): 62.76 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
new file mode 100644
index 0000000..8e6a030
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
@@ -0,0 +1,1023 @@
+WARNING 12-09 19:47:08 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:08 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:08 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 31800, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:17 [model.py:637] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:17 [model.py:1750] Using max model len 31800
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:17 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:26 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=31800, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:26 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:35389 backend=nccl
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:26 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:27 [gpu_model_runner.py:3467] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:27 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 6.84it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:00<00:00, 2.89it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:01<00:00, 2.17it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 1.96it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 2.20it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:31 [default_loader.py:308] Loading weights took 1.97 seconds
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:31 [gpu_model_runner.py:3549] Model loading took 14.9889 GiB memory and 4.010046 seconds
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:36 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/1f45c03c15/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:36 [backends.py:715] Dynamo bytecode transform time: 4.31 s
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:36 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:45 [backends.py:288] Compiling a graph for dynamic shape takes 8.33 s
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:46 [monitor.py:34] torch.compile takes 12.63 s in total
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:47 [gpu_worker.py:359] Available KV cache memory: 6.93 GiB
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:47 [kv_cache_utils.py:1286] GPU KV cache size: 56,800 tokens
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:47 [kv_cache_utils.py:1291] Maximum concurrency for 31,800 tokens per request: 1.79x
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:00, 22.54it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 23.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:00<00:00, 21.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 21.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:00<00:00, 22.13it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:00<00:00, 23.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 22.36it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 22.10it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 23.86it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 23.93it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 24.13it/s]
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:49 [gpu_model_runner.py:4466] Graph capturing finished in 2 secs, took 0.23 GiB
+[0;36m(EngineCore_DP0 pid=6154)[0;0m INFO 12-09 19:47:49 [core.py:254] init engine (profile, create kv cache, warmup model) took 17.50 seconds
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:51 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=5841)[0;0m WARNING 12-09 19:47:51 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:51 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:47:52 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=5841)[0;0m INFO: Started server process [5841]
+[0;36m(APIServer pid=5841)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=5841)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:53660 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:48:13 [loggers.py:236] Engine 000: Avg prompt throughput: 1.3 tokens/s, Avg generation throughput: 9.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:48:23 [loggers.py:236] Engine 000: Avg prompt throughput: 116.0 tokens/s, Avg generation throughput: 136.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:48:33 [loggers.py:236] Engine 000: Avg prompt throughput: 201.8 tokens/s, Avg generation throughput: 185.4 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:48:43 [loggers.py:236] Engine 000: Avg prompt throughput: 120.6 tokens/s, Avg generation throughput: 133.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:48:53 [loggers.py:236] Engine 000: Avg prompt throughput: 214.5 tokens/s, Avg generation throughput: 202.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:49:03 [loggers.py:236] Engine 000: Avg prompt throughput: 336.5 tokens/s, Avg generation throughput: 191.0 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:49:13 [loggers.py:236] Engine 000: Avg prompt throughput: 344.9 tokens/s, Avg generation throughput: 213.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60284 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:48982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:39494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:49:23 [loggers.py:236] Engine 000: Avg prompt throughput: 336.1 tokens/s, Avg generation throughput: 278.4 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:39494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:49:33 [loggers.py:236] Engine 000: Avg prompt throughput: 307.9 tokens/s, Avg generation throughput: 131.7 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:49:43 [loggers.py:236] Engine 000: Avg prompt throughput: 185.0 tokens/s, Avg generation throughput: 344.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36712 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:49:53 [loggers.py:236] Engine 000: Avg prompt throughput: 305.0 tokens/s, Avg generation throughput: 365.2 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60300 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33328 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:50:03 [loggers.py:236] Engine 000: Avg prompt throughput: 130.1 tokens/s, Avg generation throughput: 134.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:50:13 [loggers.py:236] Engine 000: Avg prompt throughput: 108.2 tokens/s, Avg generation throughput: 206.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:50:23 [loggers.py:236] Engine 000: Avg prompt throughput: 315.5 tokens/s, Avg generation throughput: 237.9 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33342 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:40678 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:50:33 [loggers.py:236] Engine 000: Avg prompt throughput: 120.9 tokens/s, Avg generation throughput: 264.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33312 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:33786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:50:43 [loggers.py:236] Engine 000: Avg prompt throughput: 87.6 tokens/s, Avg generation throughput: 167.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:50:53 [loggers.py:236] Engine 000: Avg prompt throughput: 139.1 tokens/s, Avg generation throughput: 78.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:51:03 [loggers.py:236] Engine 000: Avg prompt throughput: 141.0 tokens/s, Avg generation throughput: 245.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:60776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51092 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51098 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:51114 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49400 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:51:13 [loggers.py:236] Engine 000: Avg prompt throughput: 265.2 tokens/s, Avg generation throughput: 251.7 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:49424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:51:23 [loggers.py:236] Engine 000: Avg prompt throughput: 8.1 tokens/s, Avg generation throughput: 122.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:51:33 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:51:43 [loggers.py:236] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 137.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:51:53 [loggers.py:236] Engine 000: Avg prompt throughput: 977.2 tokens/s, Avg generation throughput: 661.5 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:52:03 [loggers.py:236] Engine 000: Avg prompt throughput: 1114.3 tokens/s, Avg generation throughput: 877.3 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:52:13 [loggers.py:236] Engine 000: Avg prompt throughput: 831.4 tokens/s, Avg generation throughput: 973.6 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:52:23 [loggers.py:236] Engine 000: Avg prompt throughput: 635.7 tokens/s, Avg generation throughput: 747.1 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:52:33 [loggers.py:236] Engine 000: Avg prompt throughput: 844.7 tokens/s, Avg generation throughput: 1004.9 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:52:43 [loggers.py:236] Engine 000: Avg prompt throughput: 944.3 tokens/s, Avg generation throughput: 848.3 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:52:53 [loggers.py:236] Engine 000: Avg prompt throughput: 619.9 tokens/s, Avg generation throughput: 750.1 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:53:03 [loggers.py:236] Engine 000: Avg prompt throughput: 928.2 tokens/s, Avg generation throughput: 1043.3 tokens/s, Running: 30 reqs, Waiting: 0 reqs, GPU KV cache usage: 16.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:53:13 [loggers.py:236] Engine 000: Avg prompt throughput: 443.1 tokens/s, Avg generation throughput: 1046.1 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54344 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:53:23 [loggers.py:236] Engine 000: Avg prompt throughput: 858.5 tokens/s, Avg generation throughput: 751.1 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36456 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:53:33 [loggers.py:236] Engine 000: Avg prompt throughput: 868.4 tokens/s, Avg generation throughput: 629.7 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:53:43 [loggers.py:236] Engine 000: Avg prompt throughput: 473.6 tokens/s, Avg generation throughput: 670.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:53:53 [loggers.py:236] Engine 000: Avg prompt throughput: 941.3 tokens/s, Avg generation throughput: 888.6 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.6%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:54:03 [loggers.py:236] Engine 000: Avg prompt throughput: 733.6 tokens/s, Avg generation throughput: 760.7 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.4%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:54:13 [loggers.py:236] Engine 000: Avg prompt throughput: 873.6 tokens/s, Avg generation throughput: 619.1 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.6%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:54:23 [loggers.py:236] Engine 000: Avg prompt throughput: 655.6 tokens/s, Avg generation throughput: 775.5 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.0%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55034 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:54:33 [loggers.py:236] Engine 000: Avg prompt throughput: 924.2 tokens/s, Avg generation throughput: 893.3 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.1%, Prefix cache hit rate: 0.1%
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:59934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:55038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:58170 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56586 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36452 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:36524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:54364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=5841)[0;0m INFO 12-09 19:54:43 [loggers.py:236] Engine 000: Avg prompt throughput: 695.3 tokens/s, Avg generation throughput: 873.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.8%, Prefix cache hit rate: 0.1%
+[rank0]:[W1209 19:55:24.307666701 ProcessGroupNCCL.cpp:1524] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
new file mode 100644
index 0000000..534f062
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 252.42282237578183,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 3.9616069204364575,
+ "tokens_per_second": 2917.0500237249767
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..14314c6
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-31ba87dc-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 183.12 \nTotal input tokens: 38756 \nTotal generated tokens: 38869 \nRequest throughput (req/s): 0.98 \nOutput token throughput (tok/s): 212.27 \nPeak output token throughput (tok/s): 609.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 423.91 \n---------------Time to First Token----------------\nMean TTFT (ms): 40.22 \nMedian TTFT (ms): 36.83 \nP99 TTFT (ms): 74.88 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 8.07 \nMedian TPOT (ms): 8.23 \nP99 TPOT (ms): 11.78 \n---------------Inter-token Latency----------------\nMean ITL (ms): 8.04 \nMedian ITL (ms): 8.17 \nP99 ITL (ms): 12.73 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..7a9f6b1
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-b94205ff-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 187.07 \nTotal input tokens: 145540 \nTotal generated tokens: 151811 \nRequest throughput (req/s): 3.85 \nOutput token throughput (tok/s): 811.50 \nPeak output token throughput (tok/s): 1549.00 \nPeak concurrent requests: 30.00 \nTotal Token throughput (tok/s): 1589.49 \n---------------Time to First Token----------------\nMean TTFT (ms): 41.02 \nMedian TTFT (ms): 36.91 \nP99 TTFT (ms): 79.84 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 14.71 \nMedian TPOT (ms): 14.74 \nP99 TPOT (ms): 17.44 \n---------------Inter-token Latency----------------\nMean ITL (ms): 14.60 \nMedian ITL (ms): 14.44 \nP99 ITL (ms): 32.49 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_server.log b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_server.log
new file mode 100644
index 0000000..179a843
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_server.log
@@ -0,0 +1,1022 @@
+WARNING 12-09 20:20:45 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:20:45 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:20:45 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 16384, 'max_num_seqs': 32}
+[0;36m(APIServer pid=19409)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:20:53 [model.py:637] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=19409)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 3.67it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 10.96it/s]
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:20:54 [model.py:1750] Using max model len 16384
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:20:54 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:20:54 [config.py:274] Overriding max cuda graph capture size to 1024 for performance.
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:02 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=16384, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:02 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:56281 backend=nccl
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:02 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:03 [gpu_model_runner.py:3467] Starting to load model openai/gpt-oss-20b...
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:03 [cuda.py:411] Using TRITON_ATTN attention backend out of potential backends: ['TRITON_ATTN']
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:03 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:03 [mxfp4.py:162] Using Marlin backend
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:01, 1.73it/s]
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:01<00:00, 1.46it/s]
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.43it/s]
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.46it/s]
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:06 [default_loader.py:308] Loading weights took 2.21 seconds
+[0;36m(EngineCore_DP0 pid=19724)[0;0m WARNING 12-09 20:21:06 [marlin_utils_fp4.py:226] Your GPU does not have native support for FP4 computation but FP4 quantization is being used. Weight-only FP4 compression will be used leveraging the Marlin kernel. This may degrade performance for compute-heavy workloads.
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:07 [gpu_model_runner.py:3549] Model loading took 13.7194 GiB memory and 3.694696 seconds
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:11 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/e1e56bd73f/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:11 [backends.py:715] Dynamo bytecode transform time: 3.61 s
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:11 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:18 [backends.py:288] Compiling a graph for dynamic shape takes 6.55 s
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:23 [monitor.py:34] torch.compile takes 10.16 s in total
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:24 [gpu_worker.py:359] Available KV cache memory: 7.11 GiB
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:24 [kv_cache_utils.py:1286] GPU KV cache size: 155,344 tokens
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:24 [kv_cache_utils.py:1291] Maximum concurrency for 16,384 tokens per request: 16.73x
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:06, 13.40it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:00<00:05, 13.62it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:00<00:05, 13.82it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:00<00:05, 13.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:00<00:05, 14.23it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:00<00:04, 14.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:00<00:04, 14.69it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:01<00:04, 14.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:01<00:04, 15.27it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:01<00:04, 15.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:01<00:03, 15.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:01<00:03, 16.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:01<00:03, 16.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:01<00:03, 16.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▌ | 30/83 [00:01<00:03, 17.51it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 39%|███▊ | 32/83 [00:02<00:02, 17.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 35/83 [00:02<00:02, 18.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 46%|████▌ | 38/83 [00:02<00:02, 19.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 49%|████▉ | 41/83 [00:02<00:02, 20.24it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 44/83 [00:02<00:01, 21.18it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 57%|█████▋ | 47/83 [00:02<00:01, 22.21it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 60%|██████ | 50/83 [00:02<00:01, 23.06it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 64%|██████▍ | 53/83 [00:02<00:01, 23.64it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 67%|██████▋ | 56/83 [00:03<00:01, 23.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 71%|███████ | 59/83 [00:03<00:00, 24.19it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 75%|███████▍ | 62/83 [00:03<00:00, 23.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 78%|███████▊ | 65/83 [00:03<00:00, 18.91it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:03<00:00, 20.31it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 86%|████████▌ | 71/83 [00:03<00:00, 21.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:03<00:00, 22.16it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 93%|█████████▎| 77/83 [00:04<00:00, 23.00it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 96%|█████████▋| 80/83 [00:04<00:00, 23.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:04<00:00, 23.82it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:04<00:00, 19.36it/s]
+[0;36m(EngineCore_DP0 pid=19724)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/7 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 29%|██▊ | 2/7 [00:00<00:00, 14.29it/s]
Capturing CUDA graphs (decode, FULL): 57%|█████▋ | 4/7 [00:00<00:00, 15.77it/s]
Capturing CUDA graphs (decode, FULL): 86%|████████▌ | 6/7 [00:00<00:00, 16.73it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 7/7 [00:00<00:00, 16.32it/s]
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:29 [gpu_model_runner.py:4466] Graph capturing finished in 5 secs, took 0.71 GiB
+[0;36m(EngineCore_DP0 pid=19724)[0;0m INFO 12-09 20:21:29 [core.py:254] init engine (profile, create kv cache, warmup model) took 22.31 seconds
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:31 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=19409)[0;0m WARNING 12-09 20:21:32 [serving_responses.py:215] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:37 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=19409)[0;0m INFO: Started server process [19409]
+[0;36m(APIServer pid=19409)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=19409)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48002 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:21:57 [loggers.py:236] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 11.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:22:07 [loggers.py:236] Engine 000: Avg prompt throughput: 114.9 tokens/s, Avg generation throughput: 220.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:22:17 [loggers.py:236] Engine 000: Avg prompt throughput: 205.8 tokens/s, Avg generation throughput: 188.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:22:27 [loggers.py:236] Engine 000: Avg prompt throughput: 188.6 tokens/s, Avg generation throughput: 146.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:22:37 [loggers.py:236] Engine 000: Avg prompt throughput: 347.7 tokens/s, Avg generation throughput: 170.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:22:47 [loggers.py:236] Engine 000: Avg prompt throughput: 176.1 tokens/s, Avg generation throughput: 219.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:48118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:22:57 [loggers.py:236] Engine 000: Avg prompt throughput: 426.9 tokens/s, Avg generation throughput: 182.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:38368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:23:07 [loggers.py:236] Engine 000: Avg prompt throughput: 296.2 tokens/s, Avg generation throughput: 238.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:23:17 [loggers.py:236] Engine 000: Avg prompt throughput: 310.8 tokens/s, Avg generation throughput: 242.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:23:27 [loggers.py:236] Engine 000: Avg prompt throughput: 172.6 tokens/s, Avg generation throughput: 374.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:33396 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:23:37 [loggers.py:236] Engine 000: Avg prompt throughput: 431.1 tokens/s, Avg generation throughput: 210.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50532 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:23:47 [loggers.py:236] Engine 000: Avg prompt throughput: 36.4 tokens/s, Avg generation throughput: 242.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:23:57 [loggers.py:236] Engine 000: Avg prompt throughput: 180.3 tokens/s, Avg generation throughput: 156.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:24:07 [loggers.py:236] Engine 000: Avg prompt throughput: 262.5 tokens/s, Avg generation throughput: 284.4 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:24:17 [loggers.py:236] Engine 000: Avg prompt throughput: 96.4 tokens/s, Avg generation throughput: 251.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56352 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:50514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:45496 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:24:27 [loggers.py:236] Engine 000: Avg prompt throughput: 68.6 tokens/s, Avg generation throughput: 58.3 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:49018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56690 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:24:37 [loggers.py:236] Engine 000: Avg prompt throughput: 135.1 tokens/s, Avg generation throughput: 226.2 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:49018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:56674 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:49018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:24:47 [loggers.py:236] Engine 000: Avg prompt throughput: 109.7 tokens/s, Avg generation throughput: 130.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:49018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:49018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:49018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60108 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:60132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:24:57 [loggers.py:236] Engine 000: Avg prompt throughput: 315.8 tokens/s, Avg generation throughput: 289.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:25:07 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 53.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:25:17 [loggers.py:236] Engine 000: Avg prompt throughput: 136.1 tokens/s, Avg generation throughput: 123.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 3.1%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:25:27 [loggers.py:236] Engine 000: Avg prompt throughput: 1057.0 tokens/s, Avg generation throughput: 664.9 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 22.7%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:25:37 [loggers.py:236] Engine 000: Avg prompt throughput: 1047.4 tokens/s, Avg generation throughput: 872.7 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 35.2%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:25:47 [loggers.py:236] Engine 000: Avg prompt throughput: 910.3 tokens/s, Avg generation throughput: 993.1 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 43.1%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:25:57 [loggers.py:236] Engine 000: Avg prompt throughput: 409.8 tokens/s, Avg generation throughput: 722.1 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 45.9%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:26:07 [loggers.py:236] Engine 000: Avg prompt throughput: 1011.1 tokens/s, Avg generation throughput: 999.9 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.8%, Prefix cache hit rate: 44.0%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:26:17 [loggers.py:236] Engine 000: Avg prompt throughput: 998.4 tokens/s, Avg generation throughput: 801.2 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 39.4%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:26:27 [loggers.py:236] Engine 000: Avg prompt throughput: 537.4 tokens/s, Avg generation throughput: 773.8 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 37.3%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:26:37 [loggers.py:236] Engine 000: Avg prompt throughput: 831.5 tokens/s, Avg generation throughput: 1031.0 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 34.4%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:26:47 [loggers.py:236] Engine 000: Avg prompt throughput: 652.9 tokens/s, Avg generation throughput: 1092.6 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 32.4%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:32800 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:26:57 [loggers.py:236] Engine 000: Avg prompt throughput: 877.1 tokens/s, Avg generation throughput: 725.7 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 30.7%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57954 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:27:07 [loggers.py:236] Engine 000: Avg prompt throughput: 914.8 tokens/s, Avg generation throughput: 692.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 28.6%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:27:17 [loggers.py:236] Engine 000: Avg prompt throughput: 507.9 tokens/s, Avg generation throughput: 649.6 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 27.6%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:27:27 [loggers.py:236] Engine 000: Avg prompt throughput: 844.8 tokens/s, Avg generation throughput: 855.0 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 26.1%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:27:37 [loggers.py:236] Engine 000: Avg prompt throughput: 894.6 tokens/s, Avg generation throughput: 735.4 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 24.6%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:27:47 [loggers.py:236] Engine 000: Avg prompt throughput: 581.1 tokens/s, Avg generation throughput: 648.9 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 23.7%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:27:57 [loggers.py:236] Engine 000: Avg prompt throughput: 732.9 tokens/s, Avg generation throughput: 834.7 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 22.7%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40826 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:28:07 [loggers.py:236] Engine 000: Avg prompt throughput: 875.2 tokens/s, Avg generation throughput: 741.3 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 21.6%
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:57976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41326 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:55076 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41362 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40824 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41346 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36726 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36738 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:36750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:59762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:40840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO: 127.0.0.1:41238 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=19409)[0;0m INFO 12-09 20:28:17 [loggers.py:236] Engine 000: Avg prompt throughput: 733.2 tokens/s, Avg generation throughput: 1047.9 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 20.7%
diff --git a/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_throughput.json
new file mode 100644
index 0000000..0f0402f
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-4090/openai_gpt-oss-20b_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 275.10345687624067,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 3.6349961260205634,
+ "tokens_per_second": 2685.5060579349843
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..13a9290
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-0abaed61-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 192.06 \nTotal input tokens: 38358 \nTotal generated tokens: 40296 \nRequest throughput (req/s): 0.94 \nOutput token throughput (tok/s): 209.81 \nPeak output token throughput (tok/s): 440.00 \nPeak concurrent requests: 13.00 \nTotal Token throughput (tok/s): 409.54 \n---------------Time to First Token----------------\nMean TTFT (ms): 51.40 \nMedian TTFT (ms): 47.89 \nP99 TTFT (ms): 93.19 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 23.06 \nMedian TPOT (ms): 23.00 \nP99 TPOT (ms): 24.12 \n---------------Inter-token Latency----------------\nMean ITL (ms): 23.10 \nMedian ITL (ms): 22.83 \nP99 ITL (ms): 28.65 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..96eddac
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-14719fc9-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 195.24 \nTotal input tokens: 146694 \nTotal generated tokens: 155585 \nRequest throughput (req/s): 3.69 \nOutput token throughput (tok/s): 796.89 \nPeak output token throughput (tok/s): 1321.00 \nPeak concurrent requests: 39.00 \nTotal Token throughput (tok/s): 1548.24 \n---------------Time to First Token----------------\nMean TTFT (ms): 53.18 \nMedian TTFT (ms): 49.55 \nP99 TTFT (ms): 101.04 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 24.41 \nMedian TPOT (ms): 24.26 \nP99 TPOT (ms): 28.40 \n---------------Inter-token Latency----------------\nMean ITL (ms): 24.37 \nMedian ITL (ms): 23.61 \nP99 ITL (ms): 56.70 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..64143d2
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1032 @@
+WARNING 12-09 19:04:10 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:10 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:10 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32768, 'max_num_seqs': 64}
+[0;36m(APIServer pid=14033)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:16 [model.py:637] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:16 [model.py:1750] Using max model len 32768
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:17 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:23 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:24 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:39463 backend=nccl
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:24 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:24 [gpu_model_runner.py:3467] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:24 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 8.25it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:00<00:00, 3.27it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:01<00:00, 2.62it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 2.33it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 2.61it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:27 [default_loader.py:308] Loading weights took 1.61 seconds
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:27 [gpu_model_runner.py:3549] Model loading took 15.3388 GiB memory and 2.767054 seconds
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:37 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/930c5e773b/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:37 [backends.py:715] Dynamo bytecode transform time: 8.89 s
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:38 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:43 [backends.py:288] Compiling a graph for dynamic shape takes 5.81 s
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:47 [monitor.py:34] torch.compile takes 14.70 s in total
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:48 [gpu_worker.py:359] Available KV cache memory: 12.32 GiB
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:48 [kv_cache_utils.py:1286] GPU KV cache size: 80,704 tokens
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:48 [kv_cache_utils.py:1291] Maximum concurrency for 32,768 tokens per request: 2.46x
+[0;36m(EngineCore_DP0 pid=14297)[0;0m 2025-12-09 19:04:48,493 - INFO - autotuner.py:256 - flashinfer.jit: [Autotuner]: Autotuning process starts ...
+[0;36m(EngineCore_DP0 pid=14297)[0;0m 2025-12-09 19:04:48,503 - INFO - autotuner.py:262 - flashinfer.jit: [Autotuner]: Autotuning process ends
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:00, 26.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 26.68it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:00<00:00, 26.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 26.77it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:00<00:00, 26.67it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:00<00:00, 26.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 26.44it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 5.90it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 16.60it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:00<00:00, 21.37it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:00<00:00, 23.85it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 21.12it/s]
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:50 [gpu_model_runner.py:4466] Graph capturing finished in 2 secs, took -0.24 GiB
+[0;36m(EngineCore_DP0 pid=14297)[0;0m INFO 12-09 19:04:50 [core.py:254] init engine (profile, create kv cache, warmup model) took 22.52 seconds
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=14033)[0;0m WARNING 12-09 19:04:52 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:04:52 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=14033)[0;0m INFO: Started server process [14033]
+[0;36m(APIServer pid=14033)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=14033)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:58726 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:05:13 [loggers.py:236] Engine 000: Avg prompt throughput: 2.4 tokens/s, Avg generation throughput: 13.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38572 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:05:23 [loggers.py:236] Engine 000: Avg prompt throughput: 114.3 tokens/s, Avg generation throughput: 131.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:05:33 [loggers.py:236] Engine 000: Avg prompt throughput: 246.5 tokens/s, Avg generation throughput: 224.3 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:05:43 [loggers.py:236] Engine 000: Avg prompt throughput: 244.2 tokens/s, Avg generation throughput: 151.0 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:05:53 [loggers.py:236] Engine 000: Avg prompt throughput: 249.4 tokens/s, Avg generation throughput: 190.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:59908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:06:03 [loggers.py:236] Engine 000: Avg prompt throughput: 342.7 tokens/s, Avg generation throughput: 217.6 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52220 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:06:13 [loggers.py:236] Engine 000: Avg prompt throughput: 284.7 tokens/s, Avg generation throughput: 202.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49714 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38568 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:06:23 [loggers.py:236] Engine 000: Avg prompt throughput: 371.6 tokens/s, Avg generation throughput: 246.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:06:33 [loggers.py:236] Engine 000: Avg prompt throughput: 247.8 tokens/s, Avg generation throughput: 191.7 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37676 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:06:43 [loggers.py:236] Engine 000: Avg prompt throughput: 169.1 tokens/s, Avg generation throughput: 363.8 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:49710 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:06:53 [loggers.py:236] Engine 000: Avg prompt throughput: 339.2 tokens/s, Avg generation throughput: 297.4 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:07:03 [loggers.py:236] Engine 000: Avg prompt throughput: 38.5 tokens/s, Avg generation throughput: 190.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:07:13 [loggers.py:236] Engine 000: Avg prompt throughput: 201.2 tokens/s, Avg generation throughput: 208.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:07:23 [loggers.py:236] Engine 000: Avg prompt throughput: 246.3 tokens/s, Avg generation throughput: 229.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:46728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:37662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:07:33 [loggers.py:236] Engine 000: Avg prompt throughput: 99.5 tokens/s, Avg generation throughput: 303.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:43692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52192 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:52234 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:07:43 [loggers.py:236] Engine 000: Avg prompt throughput: 93.6 tokens/s, Avg generation throughput: 92.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47666 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:07:53 [loggers.py:236] Engine 000: Avg prompt throughput: 120.4 tokens/s, Avg generation throughput: 166.3 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:50298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:08:03 [loggers.py:236] Engine 000: Avg prompt throughput: 161.8 tokens/s, Avg generation throughput: 219.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:51794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47696 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:36650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:36658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:36658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:50298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:50298 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:47704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:08:13 [loggers.py:236] Engine 000: Avg prompt throughput: 263.8 tokens/s, Avg generation throughput: 243.9 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:08:23 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 152.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:08:33 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 3.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:08:43 [loggers.py:236] Engine 000: Avg prompt throughput: 428.1 tokens/s, Avg generation throughput: 223.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 9.6%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:08:53 [loggers.py:236] Engine 000: Avg prompt throughput: 1264.1 tokens/s, Avg generation throughput: 747.0 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.9%, Prefix cache hit rate: 29.5%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41628 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:09:03 [loggers.py:236] Engine 000: Avg prompt throughput: 926.0 tokens/s, Avg generation throughput: 889.0 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.6%, Prefix cache hit rate: 39.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38706 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:09:13 [loggers.py:236] Engine 000: Avg prompt throughput: 672.5 tokens/s, Avg generation throughput: 1021.0 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.9%, Prefix cache hit rate: 44.4%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:09:23 [loggers.py:236] Engine 000: Avg prompt throughput: 669.6 tokens/s, Avg generation throughput: 724.3 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.4%, Prefix cache hit rate: 47.2%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:09:33 [loggers.py:236] Engine 000: Avg prompt throughput: 945.1 tokens/s, Avg generation throughput: 1039.1 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.3%, Prefix cache hit rate: 42.1%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35444 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:09:43 [loggers.py:236] Engine 000: Avg prompt throughput: 934.3 tokens/s, Avg generation throughput: 825.9 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.6%, Prefix cache hit rate: 38.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:09:53 [loggers.py:236] Engine 000: Avg prompt throughput: 503.5 tokens/s, Avg generation throughput: 782.5 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.7%, Prefix cache hit rate: 36.2%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:10:03 [loggers.py:236] Engine 000: Avg prompt throughput: 966.8 tokens/s, Avg generation throughput: 1130.1 tokens/s, Running: 31 reqs, Waiting: 0 reqs, GPU KV cache usage: 12.0%, Prefix cache hit rate: 33.0%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42012 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:10:13 [loggers.py:236] Engine 000: Avg prompt throughput: 590.6 tokens/s, Avg generation throughput: 986.8 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.8%, Prefix cache hit rate: 31.4%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41668 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38722 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:10:23 [loggers.py:236] Engine 000: Avg prompt throughput: 1023.1 tokens/s, Avg generation throughput: 774.1 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.5%, Prefix cache hit rate: 28.8%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41656 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35432 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:10:33 [loggers.py:236] Engine 000: Avg prompt throughput: 729.7 tokens/s, Avg generation throughput: 658.0 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.0%, Prefix cache hit rate: 27.3%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:10:43 [loggers.py:236] Engine 000: Avg prompt throughput: 704.5 tokens/s, Avg generation throughput: 698.1 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 8.5%, Prefix cache hit rate: 25.9%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:10:53 [loggers.py:236] Engine 000: Avg prompt throughput: 723.4 tokens/s, Avg generation throughput: 928.3 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.3%, Prefix cache hit rate: 24.8%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:11:03 [loggers.py:236] Engine 000: Avg prompt throughput: 902.9 tokens/s, Avg generation throughput: 677.2 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.2%, Prefix cache hit rate: 23.4%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:11:13 [loggers.py:236] Engine 000: Avg prompt throughput: 759.7 tokens/s, Avg generation throughput: 760.6 tokens/s, Running: 22 reqs, Waiting: 0 reqs, GPU KV cache usage: 9.1%, Prefix cache hit rate: 22.3%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:11:23 [loggers.py:236] Engine 000: Avg prompt throughput: 495.1 tokens/s, Avg generation throughput: 860.4 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.0%, Prefix cache hit rate: 21.7%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35410 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35356 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41566 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39226 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:11:33 [loggers.py:236] Engine 000: Avg prompt throughput: 1132.6 tokens/s, Avg generation throughput: 737.1 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.0%, Prefix cache hit rate: 20.3%
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38692 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39198 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:42028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:39210 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:38728 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:33768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41590 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35384 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35368 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35810 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:41576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:56798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO: 127.0.0.1:35374 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=14033)[0;0m INFO 12-09 19:11:43 [loggers.py:236] Engine 000: Avg prompt throughput: 298.0 tokens/s, Avg generation throughput: 928.9 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.9%, Prefix cache hit rate: 20.0%
+[0;36m(EngineCore_DP0 pid=14297)[0;0m /usr/lib/python3.12/multiprocessing/resource_tracker.py:123: UserWarning: resource_tracker: process died unexpectedly, relaunching. Some resources might leak.
+[0;36m(EngineCore_DP0 pid=14297)[0;0m warnings.warn('resource_tracker: process died unexpectedly, '
+Traceback (most recent call last):
+ File "/usr/lib/python3.12/multiprocessing/resource_tracker.py", line 239, in main
+ cache[rtype].remove(name)
+KeyError: '/mp-h1ustgxb'
+[rank0]:[W1209 19:13:03.210434433 ProcessGroupNCCL.cpp:1524] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..fdfacf3
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 228.03156735002995,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 4.385357745074801,
+ "tokens_per_second": 3251.0147985872827
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..f605a38
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-5741d2ff-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 182.76 \nTotal input tokens: 38358 \nTotal generated tokens: 39611 \nRequest throughput (req/s): 0.98 \nOutput token throughput (tok/s): 216.74 \nPeak output token throughput (tok/s): 789.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 426.61 \n---------------Time to First Token----------------\nMean TTFT (ms): 30.25 \nMedian TTFT (ms): 23.16 \nP99 TTFT (ms): 57.75 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 5.38 \nMedian TPOT (ms): 5.32 \nP99 TPOT (ms): 7.38 \n---------------Inter-token Latency----------------\nMean ITL (ms): 5.36 \nMedian ITL (ms): 5.29 \nP99 ITL (ms): 7.43 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..ab807da
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-4730dd07-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 183.59 \nTotal input tokens: 146694 \nTotal generated tokens: 152861 \nRequest throughput (req/s): 3.92 \nOutput token throughput (tok/s): 832.64 \nPeak output token throughput (tok/s): 1713.00 \nPeak concurrent requests: 25.00 \nTotal Token throughput (tok/s): 1631.69 \n---------------Time to First Token----------------\nMean TTFT (ms): 29.84 \nMedian TTFT (ms): 24.09 \nP99 TTFT (ms): 60.04 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 8.01 \nMedian TPOT (ms): 7.77 \nP99 TPOT (ms): 11.35 \n---------------Inter-token Latency----------------\nMean ITL (ms): 7.96 \nMedian ITL (ms): 7.28 \nP99 ITL (ms): 27.43 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
new file mode 100644
index 0000000..5575e22
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
@@ -0,0 +1,1029 @@
+WARNING 12-09 19:23:27 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:23:27 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:23:27 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'max_num_seqs': 64}
+[0;36m(APIServer pid=17193)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:23:33 [model.py:637] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:23:33 [model.py:1750] Using max model len 24576
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:23:35 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:41 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:42 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:60659 backend=nccl
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:42 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:42 [gpu_model_runner.py:3467] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:42 [compressed_tensors_wNa16.py:114] Using MarlinLinearKernel for CompressedTensorsWNA16
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:43 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:43 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:43 [compressed_tensors_moe.py:167] Using CompressedTensorsWNA16MarlinMoEMethod
+[0;36m(EngineCore_DP0 pid=17457)[0;0m WARNING 12-09 19:23:43 [compressed_tensors.py:721] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:01, 1.51it/s]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:03<00:03, 1.78s/it]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:06<00:02, 2.30s/it]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:09<00:00, 2.54s/it]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:09<00:00, 2.26s/it]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:53 [default_loader.py:308] Loading weights took 9.15 seconds
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:23:54 [gpu_model_runner.py:3549] Model loading took 15.6117 GiB memory and 11.126164 seconds
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:02 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/06a729e350/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:02 [backends.py:715] Dynamo bytecode transform time: 7.30 s
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:03 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:09 [backends.py:288] Compiling a graph for dynamic shape takes 6.57 s
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:11 [monitor.py:34] torch.compile takes 13.87 s in total
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:12 [gpu_worker.py:359] Available KV cache memory: 12.06 GiB
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:12 [kv_cache_utils.py:1286] GPU KV cache size: 131,696 tokens
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:12 [kv_cache_utils.py:1291] Maximum concurrency for 24,576 tokens per request: 5.36x
+[0;36m(EngineCore_DP0 pid=17457)[0;0m 2025-12-09 19:24:12,801 - INFO - autotuner.py:256 - flashinfer.jit: [Autotuner]: Autotuning process starts ...
+[0;36m(EngineCore_DP0 pid=17457)[0;0m 2025-12-09 19:24:12,826 - INFO - autotuner.py:262 - flashinfer.jit: [Autotuner]: Autotuning process ends
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:01, 15.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:00<00:00, 16.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 16.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 16.45it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:00<00:00, 16.73it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 17.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 17.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:00<00:00, 17.72it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:01<00:00, 17.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:01<00:00, 17.17it/s]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 5.47it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 11.49it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:00<00:00, 14.33it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:00<00:00, 15.96it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 16.86it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 17.65it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 15.41it/s]
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:15 [gpu_model_runner.py:4466] Graph capturing finished in 3 secs, took 0.08 GiB
+[0;36m(EngineCore_DP0 pid=17457)[0;0m INFO 12-09 19:24:15 [core.py:254] init engine (profile, create kv cache, warmup model) took 20.79 seconds
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=17193)[0;0m WARNING 12-09 19:24:17 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [serving_responses.py:194] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [serving_completion.py:73] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:17 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=17193)[0;0m INFO: Started server process [17193]
+[0;36m(APIServer pid=17193)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=17193)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44312 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50102 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:38 [loggers.py:236] Engine 000: Avg prompt throughput: 7.4 tokens/s, Avg generation throughput: 45.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:48 [loggers.py:236] Engine 000: Avg prompt throughput: 131.2 tokens/s, Avg generation throughput: 218.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50116 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:24:58 [loggers.py:236] Engine 000: Avg prompt throughput: 224.6 tokens/s, Avg generation throughput: 169.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:25:08 [loggers.py:236] Engine 000: Avg prompt throughput: 279.4 tokens/s, Avg generation throughput: 173.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:25:18 [loggers.py:236] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 232.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48658 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:52054 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:25:28 [loggers.py:236] Engine 000: Avg prompt throughput: 375.3 tokens/s, Avg generation throughput: 169.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50942 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:25:38 [loggers.py:236] Engine 000: Avg prompt throughput: 405.3 tokens/s, Avg generation throughput: 292.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:25:48 [loggers.py:236] Engine 000: Avg prompt throughput: 215.3 tokens/s, Avg generation throughput: 85.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:50952 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53208 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:25:58 [loggers.py:236] Engine 000: Avg prompt throughput: 298.4 tokens/s, Avg generation throughput: 362.5 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53212 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:26:08 [loggers.py:236] Engine 000: Avg prompt throughput: 225.0 tokens/s, Avg generation throughput: 387.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:26:18 [loggers.py:236] Engine 000: Avg prompt throughput: 232.7 tokens/s, Avg generation throughput: 137.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:26:28 [loggers.py:236] Engine 000: Avg prompt throughput: 41.1 tokens/s, Avg generation throughput: 209.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:26:38 [loggers.py:236] Engine 000: Avg prompt throughput: 232.2 tokens/s, Avg generation throughput: 173.7 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:26:48 [loggers.py:236] Engine 000: Avg prompt throughput: 219.2 tokens/s, Avg generation throughput: 378.3 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:53224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55508 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:26:58 [loggers.py:236] Engine 000: Avg prompt throughput: 152.8 tokens/s, Avg generation throughput: 149.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44584 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:27:08 [loggers.py:236] Engine 000: Avg prompt throughput: 70.6 tokens/s, Avg generation throughput: 31.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:27:18 [loggers.py:236] Engine 000: Avg prompt throughput: 172.9 tokens/s, Avg generation throughput: 315.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42868 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:27:28 [loggers.py:236] Engine 000: Avg prompt throughput: 130.7 tokens/s, Avg generation throughput: 128.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42894 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:42880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:44976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:27:38 [loggers.py:236] Engine 000: Avg prompt throughput: 205.6 tokens/s, Avg generation throughput: 313.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:27:48 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:27:58 [loggers.py:236] Engine 000: Avg prompt throughput: 643.5 tokens/s, Avg generation throughput: 585.6 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 13.8%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:28:08 [loggers.py:236] Engine 000: Avg prompt throughput: 1223.3 tokens/s, Avg generation throughput: 825.2 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 31.5%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:28:18 [loggers.py:236] Engine 000: Avg prompt throughput: 786.0 tokens/s, Avg generation throughput: 1045.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 39.2%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:28:28 [loggers.py:236] Engine 000: Avg prompt throughput: 684.0 tokens/s, Avg generation throughput: 784.1 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 44.7%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:28:38 [loggers.py:236] Engine 000: Avg prompt throughput: 819.3 tokens/s, Avg generation throughput: 915.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 46.1%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:28:48 [loggers.py:236] Engine 000: Avg prompt throughput: 1072.9 tokens/s, Avg generation throughput: 861.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 40.6%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:28:58 [loggers.py:236] Engine 000: Avg prompt throughput: 761.7 tokens/s, Avg generation throughput: 838.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 37.5%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:29:08 [loggers.py:236] Engine 000: Avg prompt throughput: 580.1 tokens/s, Avg generation throughput: 807.0 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 35.4%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:29:18 [loggers.py:236] Engine 000: Avg prompt throughput: 853.1 tokens/s, Avg generation throughput: 1296.5 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 32.7%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:29:28 [loggers.py:236] Engine 000: Avg prompt throughput: 677.4 tokens/s, Avg generation throughput: 744.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 30.8%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56764 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:29:38 [loggers.py:236] Engine 000: Avg prompt throughput: 1176.6 tokens/s, Avg generation throughput: 773.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 28.1%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45996 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:29:48 [loggers.py:236] Engine 000: Avg prompt throughput: 428.8 tokens/s, Avg generation throughput: 638.3 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 27.2%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:29:58 [loggers.py:236] Engine 000: Avg prompt throughput: 826.0 tokens/s, Avg generation throughput: 697.1 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 25.8%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:30:08 [loggers.py:236] Engine 000: Avg prompt throughput: 936.3 tokens/s, Avg generation throughput: 831.8 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 24.2%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56758 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:30:18 [loggers.py:236] Engine 000: Avg prompt throughput: 701.2 tokens/s, Avg generation throughput: 591.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 23.1%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:30:28 [loggers.py:236] Engine 000: Avg prompt throughput: 682.6 tokens/s, Avg generation throughput: 893.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 22.2%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:30:38 [loggers.py:236] Engine 000: Avg prompt throughput: 674.3 tokens/s, Avg generation throughput: 756.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 21.3%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:45980 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46014 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:48814 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:46018 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO 12-09 19:30:48 [loggers.py:236] Engine 000: Avg prompt throughput: 1070.2 tokens/s, Avg generation throughput: 907.6 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 20.1%
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55786 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:56762 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:55812 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=17193)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+/usr/lib/python3.12/multiprocessing/resource_tracker.py:254: UserWarning: resource_tracker: There appear to be 1 leaked semaphore objects to clean up at shutdown
+ warnings.warn('resource_tracker: There appear to be %d '
diff --git a/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
new file mode 100644
index 0000000..447986d
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 133.56136341392994,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 7.487195207051202,
+ "tokens_per_second": 5550.512371624096
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..7e1ad65
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-d6b41380-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 185.59 \nTotal input tokens: 37841 \nTotal generated tokens: 38900 \nRequest throughput (req/s): 0.97 \nOutput token throughput (tok/s): 209.60 \nPeak output token throughput (tok/s): 517.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 413.50 \n---------------Time to First Token----------------\nMean TTFT (ms): 39.73 \nMedian TTFT (ms): 32.58 \nP99 TTFT (ms): 92.75 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 11.35 \nMedian TPOT (ms): 11.23 \nP99 TPOT (ms): 13.06 \n---------------Inter-token Latency----------------\nMean ITL (ms): 11.32 \nMedian ITL (ms): 11.05 \nP99 ITL (ms): 13.41 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..aa4f809
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-10085502-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 188.38 \nTotal input tokens: 145810 \nTotal generated tokens: 152175 \nRequest throughput (req/s): 3.82 \nOutput token throughput (tok/s): 807.79 \nPeak output token throughput (tok/s): 1510.00 \nPeak concurrent requests: 29.00 \nTotal Token throughput (tok/s): 1581.79 \n---------------Time to First Token----------------\nMean TTFT (ms): 38.75 \nMedian TTFT (ms): 31.18 \nP99 TTFT (ms): 92.33 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 12.77 \nMedian TPOT (ms): 12.43 \nP99 TPOT (ms): 17.06 \n---------------Inter-token Latency----------------\nMean ITL (ms): 12.71 \nMedian ITL (ms): 11.91 \nP99 ITL (ms): 36.70 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
new file mode 100644
index 0000000..2a23afe
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
@@ -0,0 +1,1027 @@
+WARNING 12-09 18:18:16 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:16 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:16 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 65536, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:23 [model.py:637] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:23 [model.py:1750] Using max model len 65536
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:24 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:31 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=65536, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:31 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:34945 backend=nccl
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:31 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:32 [gpu_model_runner.py:3467] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:32 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:34 [weight_utils.py:487] Time spent downloading weights for meta-llama/Meta-Llama-3.1-8B-Instruct: 0.680210 seconds
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 9.79it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:00<00:00, 3.49it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:00<00:00, 2.71it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 2.47it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 2.77it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:36 [default_loader.py:308] Loading weights took 1.51 seconds
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:36 [gpu_model_runner.py:3549] Model loading took 14.9889 GiB memory and 3.826109 seconds
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:40 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/6d8c7a30a1/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:40 [backends.py:715] Dynamo bytecode transform time: 3.46 s
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:40 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:44 [backends.py:288] Compiling a graph for dynamic shape takes 3.77 s
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:45 [monitor.py:34] torch.compile takes 7.23 s in total
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:46 [gpu_worker.py:359] Available KV cache memory: 14.31 GiB
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:46 [kv_cache_utils.py:1286] GPU KV cache size: 117,232 tokens
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:46 [kv_cache_utils.py:1291] Maximum concurrency for 65,536 tokens per request: 1.79x
+[0;36m(EngineCore_DP0 pid=4670)[0;0m 2025-12-09 18:18:46,684 - INFO - autotuner.py:256 - flashinfer.jit: [Autotuner]: Autotuning process starts ...
+[0;36m(EngineCore_DP0 pid=4670)[0;0m 2025-12-09 18:18:46,695 - INFO - autotuner.py:262 - flashinfer.jit: [Autotuner]: Autotuning process ends
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:00, 25.37it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 25.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:00<00:00, 27.87it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 29.44it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:00<00:00, 31.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 29.38it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 6.08it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:00<00:00, 19.78it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 27.05it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 25.00it/s]
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:48 [gpu_model_runner.py:4466] Graph capturing finished in 2 secs, took -0.07 GiB
+[0;36m(EngineCore_DP0 pid=4670)[0;0m INFO 12-09 18:18:48 [core.py:254] init engine (profile, create kv cache, warmup model) took 11.78 seconds
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:50 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=4406)[0;0m WARNING 12-09 18:18:51 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:18:51 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=4406)[0;0m INFO: Started server process [4406]
+[0;36m(APIServer pid=4406)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=4406)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:47774 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:19:12 [loggers.py:236] Engine 000: Avg prompt throughput: 5.1 tokens/s, Avg generation throughput: 26.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43462 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:19:22 [loggers.py:236] Engine 000: Avg prompt throughput: 133.1 tokens/s, Avg generation throughput: 197.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38058 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:19:32 [loggers.py:236] Engine 000: Avg prompt throughput: 183.1 tokens/s, Avg generation throughput: 190.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:19:42 [loggers.py:236] Engine 000: Avg prompt throughput: 317.0 tokens/s, Avg generation throughput: 146.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:19:52 [loggers.py:236] Engine 000: Avg prompt throughput: 191.1 tokens/s, Avg generation throughput: 169.4 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:20:02 [loggers.py:236] Engine 000: Avg prompt throughput: 331.1 tokens/s, Avg generation throughput: 223.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:38070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:20:12 [loggers.py:236] Engine 000: Avg prompt throughput: 312.8 tokens/s, Avg generation throughput: 217.8 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50994 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:20:22 [loggers.py:236] Engine 000: Avg prompt throughput: 236.2 tokens/s, Avg generation throughput: 225.7 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:20:32 [loggers.py:236] Engine 000: Avg prompt throughput: 373.5 tokens/s, Avg generation throughput: 250.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:20:42 [loggers.py:236] Engine 000: Avg prompt throughput: 158.1 tokens/s, Avg generation throughput: 331.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:50990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45638 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:20:52 [loggers.py:236] Engine 000: Avg prompt throughput: 341.1 tokens/s, Avg generation throughput: 253.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:21:02 [loggers.py:236] Engine 000: Avg prompt throughput: 33.7 tokens/s, Avg generation throughput: 205.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:21:12 [loggers.py:236] Engine 000: Avg prompt throughput: 92.8 tokens/s, Avg generation throughput: 150.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:21:22 [loggers.py:236] Engine 000: Avg prompt throughput: 418.1 tokens/s, Avg generation throughput: 306.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45654 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33182 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48882 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:48880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:21:32 [loggers.py:236] Engine 000: Avg prompt throughput: 68.9 tokens/s, Avg generation throughput: 261.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:45632 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:54582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:21:42 [loggers.py:236] Engine 000: Avg prompt throughput: 96.3 tokens/s, Avg generation throughput: 64.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:54582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:54582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:54582 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49268 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:21:52 [loggers.py:236] Engine 000: Avg prompt throughput: 138.8 tokens/s, Avg generation throughput: 192.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:22:02 [loggers.py:236] Engine 000: Avg prompt throughput: 111.5 tokens/s, Avg generation throughput: 178.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:54290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:54290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51430 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:54290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:49280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51414 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:33166 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:22:12 [loggers.py:236] Engine 000: Avg prompt throughput: 243.1 tokens/s, Avg generation throughput: 280.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:22:22 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 30.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:22:32 [loggers.py:236] Engine 000: Avg prompt throughput: 319.1 tokens/s, Avg generation throughput: 183.8 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 7.4%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:22:42 [loggers.py:236] Engine 000: Avg prompt throughput: 909.5 tokens/s, Avg generation throughput: 756.3 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 23.5%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:22:52 [loggers.py:236] Engine 000: Avg prompt throughput: 1132.5 tokens/s, Avg generation throughput: 930.0 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.3%, Prefix cache hit rate: 36.9%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:23:02 [loggers.py:236] Engine 000: Avg prompt throughput: 779.4 tokens/s, Avg generation throughput: 946.8 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 43.5%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:23:12 [loggers.py:236] Engine 000: Avg prompt throughput: 635.8 tokens/s, Avg generation throughput: 727.9 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 48.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:23:22 [loggers.py:236] Engine 000: Avg prompt throughput: 844.9 tokens/s, Avg generation throughput: 1070.0 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 43.2%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:23:32 [loggers.py:236] Engine 000: Avg prompt throughput: 1011.7 tokens/s, Avg generation throughput: 784.2 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 38.6%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:23:42 [loggers.py:236] Engine 000: Avg prompt throughput: 552.5 tokens/s, Avg generation throughput: 738.9 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 36.4%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:23:52 [loggers.py:236] Engine 000: Avg prompt throughput: 978.7 tokens/s, Avg generation throughput: 1092.9 tokens/s, Running: 21 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.8%, Prefix cache hit rate: 33.2%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:24:02 [loggers.py:236] Engine 000: Avg prompt throughput: 418.4 tokens/s, Avg generation throughput: 1030.0 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 32.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43282 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:24:12 [loggers.py:236] Engine 000: Avg prompt throughput: 832.9 tokens/s, Avg generation throughput: 711.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 29.8%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43262 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56554 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:24:22 [loggers.py:236] Engine 000: Avg prompt throughput: 868.6 tokens/s, Avg generation throughput: 622.1 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 27.8%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:24:32 [loggers.py:236] Engine 000: Avg prompt throughput: 487.4 tokens/s, Avg generation throughput: 626.6 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 26.8%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:24:42 [loggers.py:236] Engine 000: Avg prompt throughput: 927.5 tokens/s, Avg generation throughput: 1013.2 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 25.2%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59200 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:24:52 [loggers.py:236] Engine 000: Avg prompt throughput: 733.6 tokens/s, Avg generation throughput: 654.3 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 24.0%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:25:02 [loggers.py:236] Engine 000: Avg prompt throughput: 889.4 tokens/s, Avg generation throughput: 635.6 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 22.7%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:25:12 [loggers.py:236] Engine 000: Avg prompt throughput: 639.9 tokens/s, Avg generation throughput: 816.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 21.8%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:25:22 [loggers.py:236] Engine 000: Avg prompt throughput: 926.1 tokens/s, Avg generation throughput: 852.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:56534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:44608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:44624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:44634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59216 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:51188 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:59202 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:43266 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:40008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:34286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO: 127.0.0.1:39992 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:25:32 [loggers.py:236] Engine 000: Avg prompt throughput: 693.4 tokens/s, Avg generation throughput: 932.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.2%, Prefix cache hit rate: 19.9%
+[0;36m(APIServer pid=4406)[0;0m INFO 12-09 18:25:36 [launcher.py:110] Shutting down FastAPI HTTP server.
+[rank0]:[W1209 18:28:56.997423394 ProcessGroupNCCL.cpp:1524] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
+[rank0]:[W1209 18:28:56.102110333 AllocatorConfig.cpp:28] Warning: PYTORCH_CUDA_ALLOC_CONF is deprecated, use PYTORCH_ALLOC_CONF instead (function operator())
diff --git a/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
new file mode 100644
index 0000000..3d87dc0
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 230.30906111001968,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 4.341991562035397,
+ "tokens_per_second": 3197.1386468735236
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..ec7cfb4
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-7ca49d8d-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 182.22 \nTotal input tokens: 38756 \nTotal generated tokens: 39194 \nRequest throughput (req/s): 0.99 \nOutput token throughput (tok/s): 215.09 \nPeak output token throughput (tok/s): 788.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 427.78 \n---------------Time to First Token----------------\nMean TTFT (ms): 26.17 \nMedian TTFT (ms): 21.35 \nP99 TTFT (ms): 52.39 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 5.08 \nMedian TPOT (ms): 5.07 \nP99 TPOT (ms): 7.44 \n---------------Inter-token Latency----------------\nMean ITL (ms): 5.10 \nMedian ITL (ms): 5.12 \nP99 ITL (ms): 7.77 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..81bccdf
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-1904b29f-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 183.65 \nTotal input tokens: 145540 \nTotal generated tokens: 151340 \nRequest throughput (req/s): 3.92 \nOutput token throughput (tok/s): 824.06 \nPeak output token throughput (tok/s): 1801.00 \nPeak concurrent requests: 25.00 \nTotal Token throughput (tok/s): 1616.54 \n---------------Time to First Token----------------\nMean TTFT (ms): 26.49 \nMedian TTFT (ms): 22.28 \nP99 TTFT (ms): 56.70 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 8.24 \nMedian TPOT (ms): 8.09 \nP99 TPOT (ms): 10.51 \n---------------Inter-token Latency----------------\nMean ITL (ms): 8.19 \nMedian ITL (ms): 7.72 \nP99 ITL (ms): 14.51 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_server.log b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_server.log
new file mode 100644
index 0000000..e415e3d
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_server.log
@@ -0,0 +1,1030 @@
+WARNING 12-09 18:35:12 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:12 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:12 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 32768, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=8738)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:19 [model.py:637] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=8738)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 2.56it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 7.21it/s]
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:19 [model.py:1750] Using max model len 32768
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:21 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:21 [config.py:274] Overriding max cuda graph capture size to 1024 for performance.
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:27 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:28 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.4:35797 backend=nccl
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:28 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:28 [gpu_model_runner.py:3467] Starting to load model openai/gpt-oss-20b...
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:28 [cuda.py:411] Using TRITON_ATTN attention backend out of potential backends: ['TRITON_ATTN']
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:28 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:28 [mxfp4.py:162] Using Marlin backend
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:00, 2.91it/s]
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:00<00:00, 2.23it/s]
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:01<00:00, 2.03it/s]
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:01<00:00, 2.12it/s]
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:31 [default_loader.py:308] Loading weights took 1.57 seconds
+[0;36m(EngineCore_DP0 pid=9006)[0;0m WARNING 12-09 18:35:31 [marlin_utils_fp4.py:226] Your GPU does not have native support for FP4 computation but FP4 quantization is being used. Weight-only FP4 compression will be used leveraging the Marlin kernel. This may degrade performance for compute-heavy workloads.
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:31 [gpu_model_runner.py:3549] Model loading took 13.7194 GiB memory and 2.842897 seconds
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:34 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/b0871f56e7/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:34 [backends.py:715] Dynamo bytecode transform time: 2.83 s
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:35 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:38 [backends.py:288] Compiling a graph for dynamic shape takes 2.86 s
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:39 [monitor.py:34] torch.compile takes 5.69 s in total
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:39 [gpu_worker.py:359] Available KV cache memory: 15.37 GiB
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:39 [kv_cache_utils.py:1286] GPU KV cache size: 335,664 tokens
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:39 [kv_cache_utils.py:1291] Maximum concurrency for 32,768 tokens per request: 19.20x
+[0;36m(EngineCore_DP0 pid=9006)[0;0m 2025-12-09 18:35:39,976 - INFO - autotuner.py:256 - flashinfer.jit: [Autotuner]: Autotuning process starts ...
+[0;36m(EngineCore_DP0 pid=9006)[0;0m 2025-12-09 18:35:39,993 - INFO - autotuner.py:262 - flashinfer.jit: [Autotuner]: Autotuning process ends
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:04, 17.16it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:00<00:04, 17.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:00<00:04, 17.78it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:00<00:04, 17.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:00<00:04, 18.22it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:00<00:03, 18.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:00<00:03, 18.76it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:00<00:03, 18.99it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 23%|██▎ | 19/83 [00:01<00:03, 19.79it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:01<00:02, 20.38it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 30%|███ | 25/83 [00:01<00:02, 20.95it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:01<00:02, 20.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 31/83 [00:01<00:02, 21.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 41%|████ | 34/83 [00:01<00:02, 22.33it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 45%|████▍ | 37/83 [00:01<00:01, 23.39it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 48%|████▊ | 40/83 [00:01<00:01, 24.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 52%|█████▏ | 43/83 [00:02<00:01, 25.78it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 55%|█████▌ | 46/83 [00:02<00:01, 26.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 60%|██████ | 50/83 [00:02<00:01, 28.26it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 65%|██████▌ | 54/83 [00:02<00:00, 29.33it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 70%|██████▉ | 58/83 [00:02<00:00, 29.97it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 75%|███████▍ | 62/83 [00:02<00:00, 30.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 80%|███████▉ | 66/83 [00:02<00:00, 30.30it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 70/83 [00:02<00:00, 30.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 74/83 [00:03<00:00, 30.94it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 94%|█████████▍| 78/83 [00:03<00:00, 31.21it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 99%|█████████▉| 82/83 [00:03<00:00, 31.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:03<00:00, 25.16it/s]
+[0;36m(EngineCore_DP0 pid=9006)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:01<00:10, 1.08s/it]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:01<00:07, 1.23it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:01<00:01, 3.91it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:03<00:01, 2.97it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.95it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:04<00:00, 2.70it/s]
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:47 [gpu_model_runner.py:4466] Graph capturing finished in 8 secs, took 0.33 GiB
+[0;36m(EngineCore_DP0 pid=9006)[0;0m INFO 12-09 18:35:47 [core.py:254] init engine (profile, create kv cache, warmup model) took 16.07 seconds
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:49 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=8738)[0;0m WARNING 12-09 18:35:50 [serving_responses.py:215] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:35:52 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=8738)[0;0m INFO: Started server process [8738]
+[0;36m(APIServer pid=8738)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=8738)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:60326 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:36:12 [loggers.py:236] Engine 000: Avg prompt throughput: 7.5 tokens/s, Avg generation throughput: 55.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54744 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:36:22 [loggers.py:236] Engine 000: Avg prompt throughput: 130.1 tokens/s, Avg generation throughput: 217.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:50846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:36:32 [loggers.py:236] Engine 000: Avg prompt throughput: 287.2 tokens/s, Avg generation throughput: 171.3 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:36:42 [loggers.py:236] Engine 000: Avg prompt throughput: 211.3 tokens/s, Avg generation throughput: 171.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:36:52 [loggers.py:236] Engine 000: Avg prompt throughput: 225.3 tokens/s, Avg generation throughput: 226.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37802 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:37:02 [loggers.py:236] Engine 000: Avg prompt throughput: 365.8 tokens/s, Avg generation throughput: 194.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44848 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:37:12 [loggers.py:236] Engine 000: Avg prompt throughput: 393.2 tokens/s, Avg generation throughput: 305.0 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45570 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:47622 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:37:22 [loggers.py:236] Engine 000: Avg prompt throughput: 212.4 tokens/s, Avg generation throughput: 69.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34642 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44864 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:37:32 [loggers.py:236] Engine 000: Avg prompt throughput: 291.9 tokens/s, Avg generation throughput: 367.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34608 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:37:42 [loggers.py:236] Engine 000: Avg prompt throughput: 325.2 tokens/s, Avg generation throughput: 383.0 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:37:52 [loggers.py:236] Engine 000: Avg prompt throughput: 223.0 tokens/s, Avg generation throughput: 92.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:38:02 [loggers.py:236] Engine 000: Avg prompt throughput: 44.0 tokens/s, Avg generation throughput: 226.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:38:12 [loggers.py:236] Engine 000: Avg prompt throughput: 225.8 tokens/s, Avg generation throughput: 168.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:36360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:36360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:36360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:38:22 [loggers.py:236] Engine 000: Avg prompt throughput: 214.7 tokens/s, Avg generation throughput: 375.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:36360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:36360 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:38:32 [loggers.py:236] Engine 000: Avg prompt throughput: 153.7 tokens/s, Avg generation throughput: 154.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44838 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:38:42 [loggers.py:236] Engine 000: Avg prompt throughput: 63.4 tokens/s, Avg generation throughput: 22.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:38:52 [loggers.py:236] Engine 000: Avg prompt throughput: 156.7 tokens/s, Avg generation throughput: 329.6 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34350 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:34358 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:39:02 [loggers.py:236] Engine 000: Avg prompt throughput: 91.8 tokens/s, Avg generation throughput: 110.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59184 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:37394 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:39:12 [loggers.py:236] Engine 000: Avg prompt throughput: 253.8 tokens/s, Avg generation throughput: 290.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:39:22 [loggers.py:236] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 11.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:39:32 [loggers.py:236] Engine 000: Avg prompt throughput: 827.3 tokens/s, Avg generation throughput: 654.5 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 16.9%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:39:42 [loggers.py:236] Engine 000: Avg prompt throughput: 1239.7 tokens/s, Avg generation throughput: 812.0 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 33.5%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:39:52 [loggers.py:236] Engine 000: Avg prompt throughput: 820.3 tokens/s, Avg generation throughput: 1030.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 41.1%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:40:02 [loggers.py:236] Engine 000: Avg prompt throughput: 537.6 tokens/s, Avg generation throughput: 723.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 45.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:40:12 [loggers.py:236] Engine 000: Avg prompt throughput: 983.3 tokens/s, Avg generation throughput: 967.1 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 44.9%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:40:22 [loggers.py:236] Engine 000: Avg prompt throughput: 910.8 tokens/s, Avg generation throughput: 829.7 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 40.4%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44288 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:40:32 [loggers.py:236] Engine 000: Avg prompt throughput: 670.6 tokens/s, Avg generation throughput: 806.0 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 37.7%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:40:42 [loggers.py:236] Engine 000: Avg prompt throughput: 694.0 tokens/s, Avg generation throughput: 853.9 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 35.2%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:40:52 [loggers.py:236] Engine 000: Avg prompt throughput: 709.7 tokens/s, Avg generation throughput: 1226.7 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 33.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:41:02 [loggers.py:236] Engine 000: Avg prompt throughput: 671.5 tokens/s, Avg generation throughput: 719.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 31.1%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45162 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:55174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:41:12 [loggers.py:236] Engine 000: Avg prompt throughput: 1219.4 tokens/s, Avg generation throughput: 788.3 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 28.8%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:41:22 [loggers.py:236] Engine 000: Avg prompt throughput: 408.6 tokens/s, Avg generation throughput: 620.6 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 28.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:41:32 [loggers.py:236] Engine 000: Avg prompt throughput: 792.6 tokens/s, Avg generation throughput: 766.9 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 26.6%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44308 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:41:42 [loggers.py:236] Engine 000: Avg prompt throughput: 993.7 tokens/s, Avg generation throughput: 797.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 24.8%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:41:52 [loggers.py:236] Engine 000: Avg prompt throughput: 640.4 tokens/s, Avg generation throughput: 611.5 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 23.9%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:42:02 [loggers.py:236] Engine 000: Avg prompt throughput: 660.3 tokens/s, Avg generation throughput: 826.0 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 22.9%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:42:12 [loggers.py:236] Engine 000: Avg prompt throughput: 665.7 tokens/s, Avg generation throughput: 733.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 22.0%
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44330 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44910 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59544 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59560 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59552 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59520 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39922 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:59534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39908 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39892 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:45156 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:39950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:54096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO: 127.0.0.1:44324 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8738)[0;0m INFO 12-09 18:42:22 [loggers.py:236] Engine 000: Avg prompt throughput: 1107.8 tokens/s, Avg generation throughput: 1063.8 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 20.7%
+[0;36m(EngineCore_DP0 pid=9006)[0;0m /usr/lib/python3.12/multiprocessing/resource_tracker.py:123: UserWarning: resource_tracker: process died unexpectedly, relaunching. Some resources might leak.
+[0;36m(EngineCore_DP0 pid=9006)[0;0m warnings.warn('resource_tracker: process died unexpectedly, '
+Traceback (most recent call last):
+ File "/usr/lib/python3.12/multiprocessing/resource_tracker.py", line 239, in main
+ cache[rtype].remove(name)
+KeyError: '/mp-n3z18c6r'
+[rank0]:[W1209 18:51:34.464268835 ProcessGroupNCCL.cpp:1524] Warning: WARNING: destroy_process_group() was not called before program exit, which can leak resources. For more info, please see https://pytorch.org/docs/stable/distributed.html#shutdown (function operator())
diff --git a/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_throughput.json
new file mode 100644
index 0000000..d53743f
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-5090/openai_gpt-oss-20b_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 107.1224673166871,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 9.335109851826799,
+ "tokens_per_second": 6896.704477650824
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..d29fae5
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-66e0ceca-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 187.88 \nTotal input tokens: 38358 \nTotal generated tokens: 40286 \nRequest throughput (req/s): 0.96 \nOutput token throughput (tok/s): 214.42 \nPeak output token throughput (tok/s): 460.00 \nPeak concurrent requests: 9.00 \nTotal Token throughput (tok/s): 418.59 \n---------------Time to First Token----------------\nMean TTFT (ms): 77.11 \nMedian TTFT (ms): 52.45 \nP99 TTFT (ms): 199.22 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 15.76 \nMedian TPOT (ms): 15.43 \nP99 TPOT (ms): 21.30 \n---------------Inter-token Latency----------------\nMean ITL (ms): 15.60 \nMedian ITL (ms): 14.96 \nP99 ITL (ms): 17.95 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..ff2975f
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-702c15c4-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 190.21 \nTotal input tokens: 146694 \nTotal generated tokens: 155632 \nRequest throughput (req/s): 3.79 \nOutput token throughput (tok/s): 818.19 \nPeak output token throughput (tok/s): 1520.00 \nPeak concurrent requests: 37.00 \nTotal Token throughput (tok/s): 1589.40 \n---------------Time to First Token----------------\nMean TTFT (ms): 76.15 \nMedian TTFT (ms): 45.58 \nP99 TTFT (ms): 247.34 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 19.54 \nMedian TPOT (ms): 18.83 \nP99 TPOT (ms): 35.16 \n---------------Inter-token Latency----------------\nMean ITL (ms): 19.12 \nMedian ITL (ms): 16.70 \nP99 ITL (ms): 114.66 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..8e23e99
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1024 @@
+WARNING 12-10 09:47:44 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:47:44 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:47:44 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32768, 'max_num_seqs': 64}
+[0;36m(APIServer pid=6553)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:47:54 [model.py:637] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:47:54 [model.py:1750] Using max model len 32768
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:47:54 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:03 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:03 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:36973 backend=nccl
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:03 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:04 [gpu_model_runner.py:3467] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:05 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 5.63it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:00<00:00, 2.03it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:01<00:00, 1.50it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.36it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:02<00:00, 1.54it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:09 [default_loader.py:308] Loading weights took 2.82 seconds
+[0;36m(EngineCore_DP0 pid=6815)[0;0m WARNING 12-10 09:48:09 [marlin_utils_fp8.py:98] Your GPU does not have native support for FP8 computation but FP8 quantization is being used. Weight-only FP8 compression will be used leveraging the Marlin kernel. This may degrade performance for compute-heavy workloads.
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:09 [gpu_model_runner.py:3549] Model loading took 15.3291 GiB memory and 4.337578 seconds
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:19 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/62cba2c2cc/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:19 [backends.py:715] Dynamo bytecode transform time: 8.97 s
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:20 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:27 [backends.py:288] Compiling a graph for dynamic shape takes 7.69 s
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:33 [monitor.py:34] torch.compile takes 16.66 s in total
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:35 [gpu_worker.py:359] Available KV cache memory: 19.60 GiB
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:36 [kv_cache_utils.py:1286] GPU KV cache size: 128,448 tokens
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:36 [kv_cache_utils.py:1291] Maximum concurrency for 32,768 tokens per request: 3.92x
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:01, 12.14it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 21%|██ | 4/19 [00:00<00:01, 12.85it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 13.49it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 14.25it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 53%|█████▎ | 10/19 [00:00<00:00, 15.53it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 16.32it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 16.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 16/19 [00:01<00:00, 17.38it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:01<00:00, 17.56it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:01<00:00, 16.05it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:00, 14.98it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 16.62it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 17.29it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 17.69it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:00<00:00, 18.04it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 17.67it/s]
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:39 [gpu_model_runner.py:4466] Graph capturing finished in 4 secs, took 1.09 GiB
+[0;36m(EngineCore_DP0 pid=6815)[0;0m INFO 12-10 09:48:39 [core.py:254] init engine (profile, create kv cache, warmup model) took 29.91 seconds
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:42 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=6553)[0;0m WARNING 12-10 09:48:42 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:42 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:48:43 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=6553)[0;0m INFO: Started server process [6553]
+[0;36m(APIServer pid=6553)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=6553)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:39016 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:49:14 [loggers.py:236] Engine 000: Avg prompt throughput: 84.0 tokens/s, Avg generation throughput: 132.0 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38214 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:49:24 [loggers.py:236] Engine 000: Avg prompt throughput: 137.9 tokens/s, Avg generation throughput: 181.9 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48042 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:49:34 [loggers.py:236] Engine 000: Avg prompt throughput: 212.1 tokens/s, Avg generation throughput: 167.5 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:49:44 [loggers.py:236] Engine 000: Avg prompt throughput: 295.1 tokens/s, Avg generation throughput: 182.3 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:49:54 [loggers.py:236] Engine 000: Avg prompt throughput: 259.7 tokens/s, Avg generation throughput: 209.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54010 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42498 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:50:04 [loggers.py:236] Engine 000: Avg prompt throughput: 250.8 tokens/s, Avg generation throughput: 201.5 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:40036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:40048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:40060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:50966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:40060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:50966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:54002 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:40048 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:50:14 [loggers.py:236] Engine 000: Avg prompt throughput: 488.7 tokens/s, Avg generation throughput: 300.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:50:24 [loggers.py:236] Engine 000: Avg prompt throughput: 268.5 tokens/s, Avg generation throughput: 93.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:50:34 [loggers.py:236] Engine 000: Avg prompt throughput: 185.4 tokens/s, Avg generation throughput: 403.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:38218 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:42790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:58222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:58222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47072 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:58222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:50:44 [loggers.py:236] Engine 000: Avg prompt throughput: 335.0 tokens/s, Avg generation throughput: 338.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:50:54 [loggers.py:236] Engine 000: Avg prompt throughput: 120.6 tokens/s, Avg generation throughput: 172.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:51:04 [loggers.py:236] Engine 000: Avg prompt throughput: 190.0 tokens/s, Avg generation throughput: 206.9 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:51:14 [loggers.py:236] Engine 000: Avg prompt throughput: 253.5 tokens/s, Avg generation throughput: 202.3 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:51:24 [loggers.py:236] Engine 000: Avg prompt throughput: 79.6 tokens/s, Avg generation throughput: 365.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47062 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47070 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:51:34 [loggers.py:236] Engine 000: Avg prompt throughput: 130.1 tokens/s, Avg generation throughput: 96.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:51:44 [loggers.py:236] Engine 000: Avg prompt throughput: 81.0 tokens/s, Avg generation throughput: 111.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33860 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33840 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:51:54 [loggers.py:236] Engine 000: Avg prompt throughput: 139.6 tokens/s, Avg generation throughput: 259.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33842 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:53530 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:52:04 [loggers.py:236] Engine 000: Avg prompt throughput: 275.0 tokens/s, Avg generation throughput: 206.5 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:33844 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:46126 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:52:14 [loggers.py:236] Engine 000: Avg prompt throughput: 50.4 tokens/s, Avg generation throughput: 207.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:52:24 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.2 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:52:34 [loggers.py:236] Engine 000: Avg prompt throughput: 220.0 tokens/s, Avg generation throughput: 130.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 5.1%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:52:44 [loggers.py:236] Engine 000: Avg prompt throughput: 1019.5 tokens/s, Avg generation throughput: 717.6 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 23.4%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:52:54 [loggers.py:236] Engine 000: Avg prompt throughput: 1247.7 tokens/s, Avg generation throughput: 883.5 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.3%, Prefix cache hit rate: 37.8%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:53:04 [loggers.py:236] Engine 000: Avg prompt throughput: 616.9 tokens/s, Avg generation throughput: 1019.3 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.7%, Prefix cache hit rate: 42.9%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:53:14 [loggers.py:236] Engine 000: Avg prompt throughput: 643.9 tokens/s, Avg generation throughput: 761.3 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 47.4%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:53:24 [loggers.py:236] Engine 000: Avg prompt throughput: 852.9 tokens/s, Avg generation throughput: 951.4 tokens/s, Running: 23 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.4%, Prefix cache hit rate: 43.6%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56720 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:53:34 [loggers.py:236] Engine 000: Avg prompt throughput: 1031.9 tokens/s, Avg generation throughput: 829.3 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.0%, Prefix cache hit rate: 38.9%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:53:44 [loggers.py:236] Engine 000: Avg prompt throughput: 570.4 tokens/s, Avg generation throughput: 806.1 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 36.7%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:53:54 [loggers.py:236] Engine 000: Avg prompt throughput: 984.5 tokens/s, Avg generation throughput: 969.7 tokens/s, Running: 31 reqs, Waiting: 0 reqs, GPU KV cache usage: 7.2%, Prefix cache hit rate: 33.4%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56730 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:54:04 [loggers.py:236] Engine 000: Avg prompt throughput: 523.1 tokens/s, Avg generation throughput: 1190.8 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.8%, Prefix cache hit rate: 31.9%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44856 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:54:14 [loggers.py:236] Engine 000: Avg prompt throughput: 857.9 tokens/s, Avg generation throughput: 746.3 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.8%, Prefix cache hit rate: 29.7%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:47966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56682 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56708 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:54:24 [loggers.py:236] Engine 000: Avg prompt throughput: 947.9 tokens/s, Avg generation throughput: 688.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.0%, Prefix cache hit rate: 27.6%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:54:34 [loggers.py:236] Engine 000: Avg prompt throughput: 574.3 tokens/s, Avg generation throughput: 683.6 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 26.4%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:54:44 [loggers.py:236] Engine 000: Avg prompt throughput: 800.1 tokens/s, Avg generation throughput: 874.1 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.3%, Prefix cache hit rate: 25.1%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56694 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:54:54 [loggers.py:236] Engine 000: Avg prompt throughput: 901.2 tokens/s, Avg generation throughput: 754.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 23.7%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:55:04 [loggers.py:236] Engine 000: Avg prompt throughput: 613.4 tokens/s, Avg generation throughput: 694.0 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 22.8%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:55:14 [loggers.py:236] Engine 000: Avg prompt throughput: 730.3 tokens/s, Avg generation throughput: 858.9 tokens/s, Running: 19 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 21.8%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60704 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60718 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44846 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57890 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:55:24 [loggers.py:236] Engine 000: Avg prompt throughput: 1022.7 tokens/s, Avg generation throughput: 720.1 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 20.6%
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56804 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36224 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:60698 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36236 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36222 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56702 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56672 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37950 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:35390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:35406 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:35418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:35426 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37032 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:37958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:56790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:57880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:44852 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:55052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:35418 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:36244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:55068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO: 127.0.0.1:55052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:55:34 [loggers.py:236] Engine 000: Avg prompt throughput: 511.0 tokens/s, Avg generation throughput: 1076.4 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=6553)[0;0m INFO 12-10 09:55:41 [launcher.py:110] Shutting down FastAPI HTTP server.
diff --git a/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..bee7275
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 248.4086626805365,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 4.0256245060424485,
+ "tokens_per_second": 2984.3323175624723
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..eac6a4b
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-b80b1bb5-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 184.14 \nTotal input tokens: 38358 \nTotal generated tokens: 39211 \nRequest throughput (req/s): 0.98 \nOutput token throughput (tok/s): 212.94 \nPeak output token throughput (tok/s): 560.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 421.26 \n---------------Time to First Token----------------\nMean TTFT (ms): 44.12 \nMedian TTFT (ms): 41.56 \nP99 TTFT (ms): 82.51 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 8.89 \nMedian TPOT (ms): 8.73 \nP99 TPOT (ms): 13.87 \n---------------Inter-token Latency----------------\nMean ITL (ms): 8.70 \nMedian ITL (ms): 8.44 \nP99 ITL (ms): 11.78 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..6f2703c
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-fb44348e-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 186.80 \nTotal input tokens: 146694 \nTotal generated tokens: 153333 \nRequest throughput (req/s): 3.85 \nOutput token throughput (tok/s): 820.85 \nPeak output token throughput (tok/s): 1449.00 \nPeak concurrent requests: 31.00 \nTotal Token throughput (tok/s): 1606.16 \n---------------Time to First Token----------------\nMean TTFT (ms): 44.68 \nMedian TTFT (ms): 36.77 \nP99 TTFT (ms): 89.21 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 14.46 \nMedian TPOT (ms): 14.20 \nP99 TPOT (ms): 19.40 \n---------------Inter-token Latency----------------\nMean ITL (ms): 14.28 \nMedian ITL (ms): 13.63 \nP99 ITL (ms): 44.44 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
new file mode 100644
index 0000000..8c135cc
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
@@ -0,0 +1,1026 @@
+WARNING 12-10 10:05:09 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:05:09 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:05:09 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'max_num_seqs': 64}
+[0;36m(APIServer pid=8737)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:05:19 [model.py:637] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:05:19 [model.py:1750] Using max model len 24576
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:05:19 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:28 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:29 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:47671 backend=nccl
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:29 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:29 [gpu_model_runner.py:3467] Starting to load model cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit...
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:29 [compressed_tensors_wNa16.py:114] Using MarlinLinearKernel for CompressedTensorsWNA16
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:30 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:30 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:30 [compressed_tensors_moe.py:167] Using CompressedTensorsWNA16MarlinMoEMethod
+[0;36m(EngineCore_DP0 pid=8999)[0;0m WARNING 12-10 10:05:30 [compressed_tensors.py:721] Acceleration for non-quantized schemes is not supported by Compressed Tensors. Falling back to UnquantizedLinearMethod
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:01<00:03, 1.02s/it]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:05<00:06, 3.10s/it]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:10<00:03, 3.86s/it]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:14<00:00, 4.00s/it]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:14<00:00, 3.64s/it]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:46 [default_loader.py:308] Loading weights took 14.82 seconds
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:48 [gpu_model_runner.py:3549] Model loading took 15.6116 GiB memory and 17.784780 seconds
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:58 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/cc045726be/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:58 [backends.py:715] Dynamo bytecode transform time: 10.28 s
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:05:59 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:06:06 [backends.py:288] Compiling a graph for dynamic shape takes 7.44 s
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:06:09 [monitor.py:34] torch.compile takes 17.73 s in total
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:06:10 [gpu_worker.py:359] Available KV cache memory: 19.33 GiB
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:06:11 [kv_cache_utils.py:1286] GPU KV cache size: 211,152 tokens
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:06:11 [kv_cache_utils.py:1291] Maximum concurrency for 24,576 tokens per request: 8.59x
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▌ | 1/19 [00:00<00:02, 7.98it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:01, 9.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:01, 9.87it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 37%|███▋ | 7/19 [00:00<00:01, 9.94it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:00<00:00, 10.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:01<00:00, 10.18it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 68%|██████▊ | 13/19 [00:01<00:00, 10.23it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:01<00:00, 10.25it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:01<00:00, 10.33it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:01<00:00, 10.17it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:01<00:00, 10.08it/s]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 9%|▉ | 1/11 [00:00<00:01, 8.87it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 10.00it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:00<00:00, 10.24it/s]
Capturing CUDA graphs (decode, FULL): 64%|██████▎ | 7/11 [00:00<00:00, 10.50it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 10.61it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:01<00:00, 10.74it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:01<00:00, 10.51it/s]
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:06:15 [gpu_model_runner.py:4466] Graph capturing finished in 4 secs, took 1.81 GiB
+[0;36m(EngineCore_DP0 pid=8999)[0;0m INFO 12-10 10:06:15 [core.py:254] init engine (profile, create kv cache, warmup model) took 27.34 seconds
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:17 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=8737)[0;0m WARNING 12-10 10:06:17 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:17 [serving_responses.py:194] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:17 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [serving_completion.py:73] Using default completion sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [serving_chat.py:133] Using default chat sampling params from model: {'repetition_penalty': 1.05, 'temperature': 0.7, 'top_k': 20, 'top_p': 0.8}
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:18 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=8737)[0;0m INFO: Started server process [8737]
+[0;36m(APIServer pid=8737)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=8737)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42052 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49488 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:48 [loggers.py:236] Engine 000: Avg prompt throughput: 116.7 tokens/s, Avg generation throughput: 196.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:37120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:06:58 [loggers.py:236] Engine 000: Avg prompt throughput: 206.3 tokens/s, Avg generation throughput: 161.2 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49460 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:07:08 [loggers.py:236] Engine 000: Avg prompt throughput: 111.0 tokens/s, Avg generation throughput: 175.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:07:18 [loggers.py:236] Engine 000: Avg prompt throughput: 326.7 tokens/s, Avg generation throughput: 144.8 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49470 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:07:28 [loggers.py:236] Engine 000: Avg prompt throughput: 256.1 tokens/s, Avg generation throughput: 220.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:07:38 [loggers.py:236] Engine 000: Avg prompt throughput: 387.7 tokens/s, Avg generation throughput: 158.2 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:54930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:54930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:54934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:54946 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:54934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:54934 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:36442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:07:48 [loggers.py:236] Engine 000: Avg prompt throughput: 323.8 tokens/s, Avg generation throughput: 290.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:07:58 [loggers.py:236] Engine 000: Avg prompt throughput: 302.7 tokens/s, Avg generation throughput: 119.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:08:08 [loggers.py:236] Engine 000: Avg prompt throughput: 157.0 tokens/s, Avg generation throughput: 381.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43548 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:08:18 [loggers.py:236] Engine 000: Avg prompt throughput: 362.7 tokens/s, Avg generation throughput: 301.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:08:28 [loggers.py:236] Engine 000: Avg prompt throughput: 87.1 tokens/s, Avg generation throughput: 224.8 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50932 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:08:38 [loggers.py:236] Engine 000: Avg prompt throughput: 213.8 tokens/s, Avg generation throughput: 159.7 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:08:48 [loggers.py:236] Engine 000: Avg prompt throughput: 234.9 tokens/s, Avg generation throughput: 257.9 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60242 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52616 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:08:58 [loggers.py:236] Engine 000: Avg prompt throughput: 110.0 tokens/s, Avg generation throughput: 297.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:50940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52630 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:52612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:09:08 [loggers.py:236] Engine 000: Avg prompt throughput: 94.5 tokens/s, Avg generation throughput: 60.6 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:09:18 [loggers.py:236] Engine 000: Avg prompt throughput: 81.0 tokens/s, Avg generation throughput: 186.4 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44918 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:09:28 [loggers.py:236] Engine 000: Avg prompt throughput: 139.6 tokens/s, Avg generation throughput: 218.1 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:45822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:45836 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:45850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:45850 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:09:38 [loggers.py:236] Engine 000: Avg prompt throughput: 275.0 tokens/s, Avg generation throughput: 242.1 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:44948 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:45822 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:09:48 [loggers.py:236] Engine 000: Avg prompt throughput: 50.4 tokens/s, Avg generation throughput: 136.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:09:58 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:10:08 [loggers.py:236] Engine 000: Avg prompt throughput: 644.4 tokens/s, Avg generation throughput: 524.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 13.8%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:10:18 [loggers.py:236] Engine 000: Avg prompt throughput: 1233.6 tokens/s, Avg generation throughput: 801.1 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.0%, Prefix cache hit rate: 31.7%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:10:28 [loggers.py:236] Engine 000: Avg prompt throughput: 851.4 tokens/s, Avg generation throughput: 1023.7 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 39.9%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:10:38 [loggers.py:236] Engine 000: Avg prompt throughput: 638.2 tokens/s, Avg generation throughput: 821.5 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.4%, Prefix cache hit rate: 44.9%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:10:48 [loggers.py:236] Engine 000: Avg prompt throughput: 830.1 tokens/s, Avg generation throughput: 850.4 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 45.8%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:10:58 [loggers.py:236] Engine 000: Avg prompt throughput: 1031.1 tokens/s, Avg generation throughput: 891.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 40.6%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:11:08 [loggers.py:236] Engine 000: Avg prompt throughput: 761.8 tokens/s, Avg generation throughput: 847.3 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 37.5%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:11:18 [loggers.py:236] Engine 000: Avg prompt throughput: 601.3 tokens/s, Avg generation throughput: 788.7 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 35.3%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:11:28 [loggers.py:236] Engine 000: Avg prompt throughput: 834.6 tokens/s, Avg generation throughput: 1222.6 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.4%, Prefix cache hit rate: 32.7%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:11:38 [loggers.py:236] Engine 000: Avg prompt throughput: 674.9 tokens/s, Avg generation throughput: 844.7 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 30.8%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:60680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46446 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46580 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:11:48 [loggers.py:236] Engine 000: Avg prompt throughput: 1199.9 tokens/s, Avg generation throughput: 715.1 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 28.6%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41774 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:11:58 [loggers.py:236] Engine 000: Avg prompt throughput: 405.5 tokens/s, Avg generation throughput: 686.0 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 27.7%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:12:08 [loggers.py:236] Engine 000: Avg prompt throughput: 825.9 tokens/s, Avg generation throughput: 733.1 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 26.3%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:12:18 [loggers.py:236] Engine 000: Avg prompt throughput: 936.3 tokens/s, Avg generation throughput: 823.5 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 24.7%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:12:28 [loggers.py:236] Engine 000: Avg prompt throughput: 701.3 tokens/s, Avg generation throughput: 608.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 23.6%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:12:38 [loggers.py:236] Engine 000: Avg prompt throughput: 682.7 tokens/s, Avg generation throughput: 867.8 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 22.6%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:12:48 [loggers.py:236] Engine 000: Avg prompt throughput: 674.3 tokens/s, Avg generation throughput: 777.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 21.8%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:33008 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43972 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46450 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41760 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46588 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46492 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46478 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:43960 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46468 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46514 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48366 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48370 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48398 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:41780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:12:58 [loggers.py:236] Engine 000: Avg prompt throughput: 1074.7 tokens/s, Avg generation throughput: 830.4 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.5%, Prefix cache hit rate: 20.5%
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48382 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:48416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:42336 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:49364 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO: 127.0.0.1:46442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=8737)[0;0m INFO 12-10 10:13:06 [launcher.py:110] Shutting down FastAPI HTTP server.
diff --git a/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
new file mode 100644
index 0000000..f02bddf
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 199.3360565919429,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 5.016653871341908,
+ "tokens_per_second": 3719.016081057382
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..7f1bd19
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-4225f4cd-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 187.19 \nTotal input tokens: 37841 \nTotal generated tokens: 38830 \nRequest throughput (req/s): 0.96 \nOutput token throughput (tok/s): 207.43 \nPeak output token throughput (tok/s): 490.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 409.58 \n---------------Time to First Token----------------\nMean TTFT (ms): 45.84 \nMedian TTFT (ms): 37.87 \nP99 TTFT (ms): 101.24 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 14.35 \nMedian TPOT (ms): 14.27 \nP99 TPOT (ms): 16.03 \n---------------Inter-token Latency----------------\nMean ITL (ms): 14.28 \nMedian ITL (ms): 14.04 \nP99 ITL (ms): 16.81 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..5346632
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='meta-llama/Meta-Llama-3.1-8B-Instruct', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-4ca67e7d-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 190.55 \nTotal input tokens: 145810 \nTotal generated tokens: 152171 \nRequest throughput (req/s): 3.78 \nOutput token throughput (tok/s): 798.60 \nPeak output token throughput (tok/s): 1492.00 \nPeak concurrent requests: 35.00 \nTotal Token throughput (tok/s): 1563.82 \n---------------Time to First Token----------------\nMean TTFT (ms): 44.19 \nMedian TTFT (ms): 37.54 \nP99 TTFT (ms): 101.54 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 16.05 \nMedian TPOT (ms): 15.69 \nP99 TPOT (ms): 20.99 \n---------------Inter-token Latency----------------\nMean ITL (ms): 15.95 \nMedian ITL (ms): 15.02 \nP99 ITL (ms): 41.81 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
new file mode 100644
index 0000000..354c6bc
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_server.log
@@ -0,0 +1,1022 @@
+WARNING 12-10 09:24:11 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:11 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:11 [utils.py:253] non-default args: {'model_tag': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'host': '127.0.0.1', 'model': 'meta-llama/Meta-Llama-3.1-8B-Instruct', 'max_model_len': 65536, 'gpu_memory_utilization': 0.95, 'max_num_seqs': 64}
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:21 [model.py:637] Resolved architecture: LlamaForCausalLM
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:21 [model.py:1750] Using max model len 65536
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:21 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:30 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='meta-llama/Meta-Llama-3.1-8B-Instruct', speculative_config=None, tokenizer='meta-llama/Meta-Llama-3.1-8B-Instruct', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=False, dtype=torch.bfloat16, max_seq_len=65536, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=meta-llama/Meta-Llama-3.1-8B-Instruct, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:30 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:46573 backend=nccl
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:30 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:31 [gpu_model_runner.py:3467] Starting to load model meta-llama/Meta-Llama-3.1-8B-Instruct...
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:32 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 4.87it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:01<00:01, 1.62it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:02<00:00, 1.18it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.08it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:03<00:00, 1.22it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:37 [default_loader.py:308] Loading weights took 3.50 seconds
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:38 [gpu_model_runner.py:3549] Model loading took 14.9889 GiB memory and 5.683476 seconds
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:44 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/d7f4cb5e01/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:44 [backends.py:715] Dynamo bytecode transform time: 5.00 s
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:45 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:49 [backends.py:288] Compiling a graph for dynamic shape takes 3.82 s
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:50 [monitor.py:34] torch.compile takes 8.82 s in total
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:52 [gpu_worker.py:359] Available KV cache memory: 21.99 GiB
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:52 [kv_cache_utils.py:1286] GPU KV cache size: 180,128 tokens
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:52 [kv_cache_utils.py:1291] Maximum concurrency for 65,536 tokens per request: 2.75x
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 11%|█ | 2/19 [00:00<00:00, 17.69it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 26%|██▋ | 5/19 [00:00<00:00, 19.57it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 42%|████▏ | 8/19 [00:00<00:00, 20.71it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 11/19 [00:00<00:00, 21.70it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 74%|███████▎ | 14/19 [00:00<00:00, 22.92it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 89%|████████▉ | 17/19 [00:00<00:00, 24.08it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 22.20it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:00, 19.47it/s]
Capturing CUDA graphs (decode, FULL): 45%|████▌ | 5/11 [00:00<00:00, 23.29it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 24.79it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 25.49it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 24.66it/s]
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:55 [gpu_model_runner.py:4466] Graph capturing finished in 3 secs, took 0.84 GiB
+[0;36m(EngineCore_DP0 pid=3002)[0;0m INFO 12-10 09:24:55 [core.py:254] init engine (profile, create kv cache, warmup model) took 16.62 seconds
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:58 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=2740)[0;0m WARNING 12-10 09:24:58 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:58 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_p': 0.9}
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:24:59 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=2740)[0;0m INFO: Started server process [2740]
+[0;36m(APIServer pid=2740)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=2740)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48186 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:25:30 [loggers.py:236] Engine 000: Avg prompt throughput: 117.3 tokens/s, Avg generation throughput: 142.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55556 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55562 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49540 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:25:40 [loggers.py:236] Engine 000: Avg prompt throughput: 201.8 tokens/s, Avg generation throughput: 170.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:25:50 [loggers.py:236] Engine 000: Avg prompt throughput: 107.1 tokens/s, Avg generation throughput: 139.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55546 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:26:00 [loggers.py:236] Engine 000: Avg prompt throughput: 220.0 tokens/s, Avg generation throughput: 201.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:26:10 [loggers.py:236] Engine 000: Avg prompt throughput: 285.9 tokens/s, Avg generation throughput: 188.6 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:26:20 [loggers.py:236] Engine 000: Avg prompt throughput: 301.1 tokens/s, Avg generation throughput: 209.9 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55536 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57372 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49458 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45662 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:26:30 [loggers.py:236] Engine 000: Avg prompt throughput: 438.5 tokens/s, Avg generation throughput: 304.5 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:26:40 [loggers.py:236] Engine 000: Avg prompt throughput: 280.0 tokens/s, Avg generation throughput: 87.5 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:26:50 [loggers.py:236] Engine 000: Avg prompt throughput: 198.2 tokens/s, Avg generation throughput: 368.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44442 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44386 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44402 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:27:00 [loggers.py:236] Engine 000: Avg prompt throughput: 288.1 tokens/s, Avg generation throughput: 370.2 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.9%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44412 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:34818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:27:10 [loggers.py:236] Engine 000: Avg prompt throughput: 159.8 tokens/s, Avg generation throughput: 125.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:34818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:27:20 [loggers.py:236] Engine 000: Avg prompt throughput: 33.4 tokens/s, Avg generation throughput: 210.7 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:34818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:34818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:27:30 [loggers.py:236] Engine 000: Avg prompt throughput: 321.8 tokens/s, Avg generation throughput: 234.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:34818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:44428 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:27:40 [loggers.py:236] Engine 000: Avg prompt throughput: 187.7 tokens/s, Avg generation throughput: 279.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:57388 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45650 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:34818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:27:50 [loggers.py:236] Engine 000: Avg prompt throughput: 91.3 tokens/s, Avg generation throughput: 168.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:28:00 [loggers.py:236] Engine 000: Avg prompt throughput: 133.9 tokens/s, Avg generation throughput: 68.8 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51524 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:28:10 [loggers.py:236] Engine 000: Avg prompt throughput: 80.2 tokens/s, Avg generation throughput: 255.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51512 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:59742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:59748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:59748 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:59754 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:59768 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:28:20 [loggers.py:236] Engine 000: Avg prompt throughput: 330.3 tokens/s, Avg generation throughput: 233.8 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53634 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:59742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:28:30 [loggers.py:236] Engine 000: Avg prompt throughput: 8.9 tokens/s, Avg generation throughput: 136.4 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:28:40 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:28:50 [loggers.py:236] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 138.8 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 5.1%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:29:00 [loggers.py:236] Engine 000: Avg prompt throughput: 977.3 tokens/s, Avg generation throughput: 702.9 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.1%, Prefix cache hit rate: 23.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:29:10 [loggers.py:236] Engine 000: Avg prompt throughput: 1114.3 tokens/s, Avg generation throughput: 886.0 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.9%, Prefix cache hit rate: 36.4%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:29:20 [loggers.py:236] Engine 000: Avg prompt throughput: 824.8 tokens/s, Avg generation throughput: 992.6 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 43.5%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:29:30 [loggers.py:236] Engine 000: Avg prompt throughput: 528.8 tokens/s, Avg generation throughput: 739.8 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 47.2%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:29:40 [loggers.py:236] Engine 000: Avg prompt throughput: 904.2 tokens/s, Avg generation throughput: 1004.7 tokens/s, Running: 18 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 43.5%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:29:50 [loggers.py:236] Engine 000: Avg prompt throughput: 998.8 tokens/s, Avg generation throughput: 812.2 tokens/s, Running: 16 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 38.9%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:30:00 [loggers.py:236] Engine 000: Avg prompt throughput: 547.5 tokens/s, Avg generation throughput: 743.1 tokens/s, Running: 15 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.4%, Prefix cache hit rate: 36.7%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:30:10 [loggers.py:236] Engine 000: Avg prompt throughput: 902.9 tokens/s, Avg generation throughput: 1072.2 tokens/s, Running: 24 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.9%, Prefix cache hit rate: 33.6%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:30:20 [loggers.py:236] Engine 000: Avg prompt throughput: 541.0 tokens/s, Avg generation throughput: 1051.2 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 32.0%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50392 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:30:30 [loggers.py:236] Engine 000: Avg prompt throughput: 820.5 tokens/s, Avg generation throughput: 727.3 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.9%, Prefix cache hit rate: 29.9%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53120 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50424 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:51390 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60986 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50420 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:30:40 [loggers.py:236] Engine 000: Avg prompt throughput: 902.7 tokens/s, Avg generation throughput: 642.3 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.2%, Prefix cache hit rate: 28.4%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:30:50 [loggers.py:236] Engine 000: Avg prompt throughput: 477.5 tokens/s, Avg generation throughput: 641.1 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.3%, Prefix cache hit rate: 27.4%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:31:00 [loggers.py:236] Engine 000: Avg prompt throughput: 941.3 tokens/s, Avg generation throughput: 918.7 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.3%, Prefix cache hit rate: 25.7%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60976 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48782 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:31:10 [loggers.py:236] Engine 000: Avg prompt throughput: 689.2 tokens/s, Avg generation throughput: 763.4 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.7%, Prefix cache hit rate: 24.6%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:31:20 [loggers.py:236] Engine 000: Avg prompt throughput: 886.5 tokens/s, Avg generation throughput: 589.1 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.3%, Prefix cache hit rate: 23.2%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:31:30 [loggers.py:236] Engine 000: Avg prompt throughput: 686.3 tokens/s, Avg generation throughput: 807.1 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.6%, Prefix cache hit rate: 22.3%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47670 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:31:40 [loggers.py:236] Engine 000: Avg prompt throughput: 906.9 tokens/s, Avg generation throughput: 878.0 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.1%, Prefix cache hit rate: 21.1%
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:55152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45924 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:45940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60962 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:39334 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50408 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50404 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:50416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:42618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53142 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53130 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:48794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:53150 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:47680 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:49612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:42624 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:42626 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO: 127.0.0.1:60988 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:31:50 [loggers.py:236] Engine 000: Avg prompt throughput: 713.6 tokens/s, Avg generation throughput: 907.7 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 20.3%
+[0;36m(APIServer pid=2740)[0;0m INFO 12-10 09:31:57 [launcher.py:110] Shutting down FastAPI HTTP server.
diff --git a/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
new file mode 100644
index 0000000..9b1d78a
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/meta-llama_Meta-Llama-3.1-8B-Instruct_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 195.21851312741637,
+ "num_requests": 1000,
+ "total_num_tokens": 736330,
+ "requests_per_second": 5.122464995660089,
+ "tokens_per_second": 3771.8246502543934
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..2662344
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-85d6e157-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 183.05 \nTotal input tokens: 38756 \nTotal generated tokens: 38779 \nRequest throughput (req/s): 0.98 \nOutput token throughput (tok/s): 211.85 \nPeak output token throughput (tok/s): 676.00 \nPeak concurrent requests: 8.00 \nTotal Token throughput (tok/s): 423.58 \n---------------Time to First Token----------------\nMean TTFT (ms): 36.19 \nMedian TTFT (ms): 30.20 \nP99 TTFT (ms): 83.37 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 7.40 \nMedian TPOT (ms): 7.32 \nP99 TPOT (ms): 10.60 \n---------------Inter-token Latency----------------\nMean ITL (ms): 7.34 \nMedian ITL (ms): 7.07 \nP99 ITL (ms): 10.69 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..f8b7077
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='openai/gpt-oss-20b', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-584a872e-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 185.93 \nTotal input tokens: 145540 \nTotal generated tokens: 151690 \nRequest throughput (req/s): 3.87 \nOutput token throughput (tok/s): 815.86 \nPeak output token throughput (tok/s): 1630.00 \nPeak concurrent requests: 28.00 \nTotal Token throughput (tok/s): 1598.63 \n---------------Time to First Token----------------\nMean TTFT (ms): 38.36 \nMedian TTFT (ms): 31.91 \nP99 TTFT (ms): 78.67 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 12.67 \nMedian TPOT (ms): 12.82 \nP99 TPOT (ms): 16.44 \n---------------Inter-token Latency----------------\nMean ITL (ms): 12.51 \nMedian ITL (ms): 12.62 \nP99 ITL (ms): 31.79 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_server.log b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_server.log
new file mode 100644
index 0000000..fb0eaef
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_server.log
@@ -0,0 +1,1023 @@
+WARNING 12-10 10:51:19 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:51:19 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:51:19 [utils.py:253] non-default args: {'model_tag': 'openai/gpt-oss-20b', 'host': '127.0.0.1', 'model': 'openai/gpt-oss-20b', 'trust_remote_code': True, 'max_model_len': 24000, 'max_num_seqs': 64}
+[0;36m(APIServer pid=15750)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:51:29 [model.py:637] Resolved architecture: GptOssForCausalLM
+[0;36m(APIServer pid=15750)[0;0m
Parse safetensors files: 0%| | 0/3 [00:00, ?it/s]
Parse safetensors files: 33%|███▎ | 1/3 [00:00<00:00, 4.10it/s]
Parse safetensors files: 100%|██████████| 3/3 [00:00<00:00, 9.27it/s]
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:51:30 [model.py:1750] Using max model len 24000
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:51:30 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:51:30 [config.py:274] Overriding max cuda graph capture size to 1024 for performance.
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:39 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='openai/gpt-oss-20b', speculative_config=None, tokenizer='openai/gpt-oss-20b', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24000, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=mxfp4, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='openai_gptoss', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=openai/gpt-oss-20b, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128, 136, 144, 152, 160, 168, 176, 184, 192, 200, 208, 216, 224, 232, 240, 248, 256, 272, 288, 304, 320, 336, 352, 368, 384, 400, 416, 432, 448, 464, 480, 496, 512, 528, 544, 560, 576, 592, 608, 624, 640, 656, 672, 688, 704, 720, 736, 752, 768, 784, 800, 816, 832, 848, 864, 880, 896, 912, 928, 944, 960, 976, 992, 1008, 1024], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 1024, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:40 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.2:59277 backend=nccl
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:40 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:41 [gpu_model_runner.py:3467] Starting to load model openai/gpt-oss-20b...
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:41 [cuda.py:411] Using TRITON_ATTN attention backend out of potential backends: ['TRITON_ATTN']
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:41 [layer.py:379] Enabled separate cuda stream for MoE shared_experts
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:41 [mxfp4.py:162] Using Marlin backend
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/3 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
Loading safetensors checkpoint shards: 33% Completed | 1/3 [00:00<00:01, 1.81it/s]
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
Loading safetensors checkpoint shards: 67% Completed | 2/3 [00:01<00:00, 1.52it/s]
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.45it/s]
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 3/3 [00:02<00:00, 1.49it/s]
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:44 [default_loader.py:308] Loading weights took 2.14 seconds
+[0;36m(EngineCore_DP0 pid=16016)[0;0m WARNING 12-10 10:51:44 [marlin_utils_fp4.py:226] Your GPU does not have native support for FP4 computation but FP4 quantization is being used. Weight-only FP4 compression will be used leveraging the Marlin kernel. This may degrade performance for compute-heavy workloads.
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:45 [gpu_model_runner.py:3549] Model loading took 13.7025 GiB memory and 4.053831 seconds
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:50 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/09d46d6ca4/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:50 [backends.py:715] Dynamo bytecode transform time: 3.97 s
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:50 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:54 [backends.py:288] Compiling a graph for dynamic shape takes 3.92 s
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:51:58 [monitor.py:34] torch.compile takes 7.89 s in total
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:52:00 [gpu_worker.py:359] Available KV cache memory: 21.09 GiB
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:52:00 [kv_cache_utils.py:1286] GPU KV cache size: 460,688 tokens
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:52:00 [kv_cache_utils.py:1291] Maximum concurrency for 24,000 tokens per request: 35.18x
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/83 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 1%| | 1/83 [00:00<00:11, 6.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 2%|▏ | 2/83 [00:00<00:10, 7.84it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 5%|▍ | 4/83 [00:00<00:08, 9.85it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 7%|▋ | 6/83 [00:00<00:07, 10.93it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 10%|▉ | 8/83 [00:00<00:06, 11.72it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 12%|█▏ | 10/83 [00:00<00:05, 12.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 14%|█▍ | 12/83 [00:01<00:05, 12.80it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 17%|█▋ | 14/83 [00:01<00:05, 13.21it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 19%|█▉ | 16/83 [00:01<00:04, 13.52it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 22%|██▏ | 18/83 [00:01<00:04, 13.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 24%|██▍ | 20/83 [00:01<00:04, 14.11it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 27%|██▋ | 22/83 [00:01<00:04, 14.41it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 29%|██▉ | 24/83 [00:01<00:04, 14.63it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 31%|███▏ | 26/83 [00:01<00:03, 15.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 34%|███▎ | 28/83 [00:02<00:03, 15.28it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 36%|███▌ | 30/83 [00:02<00:03, 15.82it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 39%|███▊ | 32/83 [00:02<00:03, 16.12it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 41%|████ | 34/83 [00:02<00:02, 16.59it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 43%|████▎ | 36/83 [00:02<00:02, 17.04it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 46%|████▌ | 38/83 [00:02<00:02, 17.47it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 48%|████▊ | 40/83 [00:02<00:02, 17.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 51%|█████ | 42/83 [00:02<00:02, 18.36it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 54%|█████▍ | 45/83 [00:03<00:01, 19.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 58%|█████▊ | 48/83 [00:03<00:01, 19.43it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 61%|██████▏ | 51/83 [00:03<00:01, 19.72it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 65%|██████▌ | 54/83 [00:03<00:01, 19.85it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 67%|██████▋ | 56/83 [00:03<00:01, 19.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 70%|██████▉ | 58/83 [00:03<00:01, 19.83it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 72%|███████▏ | 60/83 [00:03<00:01, 19.77it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 76%|███████▌ | 63/83 [00:03<00:01, 19.96it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 78%|███████▊ | 65/83 [00:04<00:00, 19.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 82%|████████▏ | 68/83 [00:04<00:00, 19.66it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 84%|████████▍ | 70/83 [00:04<00:00, 19.68it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 88%|████████▊ | 73/83 [00:04<00:00, 19.81it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 92%|█████████▏| 76/83 [00:04<00:00, 20.03it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 94%|█████████▍| 78/83 [00:04<00:00, 19.90it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 98%|█████████▊| 81/83 [00:04<00:00, 20.08it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 83/83 [00:04<00:00, 16.83it/s]
+[0;36m(EngineCore_DP0 pid=16016)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 18%|█▊ | 2/11 [00:00<00:00, 13.38it/s]
Capturing CUDA graphs (decode, FULL): 36%|███▋ | 4/11 [00:00<00:00, 16.48it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 17.78it/s]
Capturing CUDA graphs (decode, FULL): 73%|███████▎ | 8/11 [00:00<00:00, 17.45it/s]
Capturing CUDA graphs (decode, FULL): 91%|█████████ | 10/11 [00:00<00:00, 17.73it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 17.09it/s]
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:52:07 [gpu_model_runner.py:4466] Graph capturing finished in 7 secs, took 2.67 GiB
+[0;36m(EngineCore_DP0 pid=16016)[0;0m INFO 12-10 10:52:07 [core.py:254] init engine (profile, create kv cache, warmup model) took 21.87 seconds
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:11 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=15750)[0;0m WARNING 12-10 10:52:11 [serving_responses.py:215] For gpt-oss, we ignore --enable-auto-tool-choice and always enable tool use.
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /openapi.json, Methods: HEAD, GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /docs, Methods: HEAD, GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: HEAD, GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /redoc, Methods: HEAD, GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:13 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=15750)[0;0m INFO: Started server process [15750]
+[0;36m(APIServer pid=15750)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=15750)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:38740 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:34 [loggers.py:236] Engine 000: Avg prompt throughput: 1.2 tokens/s, Avg generation throughput: 6.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42818 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:44 [loggers.py:236] Engine 000: Avg prompt throughput: 114.9 tokens/s, Avg generation throughput: 220.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42138 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42794 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:52:54 [loggers.py:236] Engine 000: Avg prompt throughput: 205.8 tokens/s, Avg generation throughput: 175.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:53:04 [loggers.py:236] Engine 000: Avg prompt throughput: 188.6 tokens/s, Avg generation throughput: 159.0 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:53:14 [loggers.py:236] Engine 000: Avg prompt throughput: 318.1 tokens/s, Avg generation throughput: 145.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:53:24 [loggers.py:236] Engine 000: Avg prompt throughput: 205.8 tokens/s, Avg generation throughput: 223.3 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:42808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41376 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:60044 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:53:34 [loggers.py:236] Engine 000: Avg prompt throughput: 425.6 tokens/s, Avg generation throughput: 180.4 tokens/s, Running: 3 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41378 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:53:44 [loggers.py:236] Engine 000: Avg prompt throughput: 247.7 tokens/s, Avg generation throughput: 260.9 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:53:54 [loggers.py:236] Engine 000: Avg prompt throughput: 336.2 tokens/s, Avg generation throughput: 217.4 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40152 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:54:04 [loggers.py:236] Engine 000: Avg prompt throughput: 169.5 tokens/s, Avg generation throughput: 361.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53244 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40160 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:53256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:54:14 [loggers.py:236] Engine 000: Avg prompt throughput: 456.9 tokens/s, Avg generation throughput: 252.5 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:54:24 [loggers.py:236] Engine 000: Avg prompt throughput: 32.8 tokens/s, Avg generation throughput: 232.9 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40148 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:54:34 [loggers.py:236] Engine 000: Avg prompt throughput: 185.5 tokens/s, Avg generation throughput: 150.7 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:50614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:50614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:50614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:33504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:33504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:50614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:54:44 [loggers.py:236] Engine 000: Avg prompt throughput: 255.5 tokens/s, Avg generation throughput: 282.2 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:50614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:50614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:54:54 [loggers.py:236] Engine 000: Avg prompt throughput: 68.4 tokens/s, Avg generation throughput: 222.4 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40134 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:50614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40174 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:40128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:55:04 [loggers.py:236] Engine 000: Avg prompt throughput: 103.6 tokens/s, Avg generation throughput: 65.3 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:37454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:37454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:37454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:37466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:37454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:55:14 [loggers.py:236] Engine 000: Avg prompt throughput: 110.7 tokens/s, Avg generation throughput: 199.9 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49296 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:37466 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49276 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:37454 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:55:24 [loggers.py:236] Engine 000: Avg prompt throughput: 134.1 tokens/s, Avg generation throughput: 175.2 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59294 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:49290 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59258 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:55:34 [loggers.py:236] Engine 000: Avg prompt throughput: 269.3 tokens/s, Avg generation throughput: 288.0 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59272 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:55:44 [loggers.py:236] Engine 000: Avg prompt throughput: 46.5 tokens/s, Avg generation throughput: 69.8 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:55:54 [loggers.py:236] Engine 000: Avg prompt throughput: 41.0 tokens/s, Avg generation throughput: 41.1 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.1%, Prefix cache hit rate: 0.9%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:56:04 [loggers.py:236] Engine 000: Avg prompt throughput: 820.4 tokens/s, Avg generation throughput: 648.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 17.4%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:56:14 [loggers.py:236] Engine 000: Avg prompt throughput: 1263.3 tokens/s, Avg generation throughput: 818.7 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 34.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:56:24 [loggers.py:236] Engine 000: Avg prompt throughput: 817.9 tokens/s, Avg generation throughput: 996.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 41.5%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:56:34 [loggers.py:236] Engine 000: Avg prompt throughput: 607.5 tokens/s, Avg generation throughput: 769.4 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 45.9%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:56:44 [loggers.py:236] Engine 000: Avg prompt throughput: 876.1 tokens/s, Avg generation throughput: 955.6 tokens/s, Running: 17 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.1%, Prefix cache hit rate: 44.8%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:56:54 [loggers.py:236] Engine 000: Avg prompt throughput: 988.4 tokens/s, Avg generation throughput: 820.6 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 40.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45796 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:57:04 [loggers.py:236] Engine 000: Avg prompt throughput: 591.8 tokens/s, Avg generation throughput: 801.7 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 37.6%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:57:14 [loggers.py:236] Engine 000: Avg prompt throughput: 755.7 tokens/s, Avg generation throughput: 894.8 tokens/s, Running: 20 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 35.0%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:57:24 [loggers.py:236] Engine 000: Avg prompt throughput: 752.7 tokens/s, Avg generation throughput: 1174.4 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 32.7%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:57:34 [loggers.py:236] Engine 000: Avg prompt throughput: 769.2 tokens/s, Avg generation throughput: 719.4 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.4%, Prefix cache hit rate: 30.6%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41612 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55256 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45780 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:57:44 [loggers.py:236] Engine 000: Avg prompt throughput: 1069.9 tokens/s, Avg generation throughput: 758.1 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 28.7%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45830 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:57:54 [loggers.py:236] Engine 000: Avg prompt throughput: 353.8 tokens/s, Avg generation throughput: 647.2 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 27.9%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:58:04 [loggers.py:236] Engine 000: Avg prompt throughput: 796.9 tokens/s, Avg generation throughput: 783.2 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.7%, Prefix cache hit rate: 26.5%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:58:14 [loggers.py:236] Engine 000: Avg prompt throughput: 1046.2 tokens/s, Avg generation throughput: 806.4 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 24.7%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:58:24 [loggers.py:236] Engine 000: Avg prompt throughput: 569.8 tokens/s, Avg generation throughput: 617.2 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.6%, Prefix cache hit rate: 23.9%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:58:34 [loggers.py:236] Engine 000: Avg prompt throughput: 727.4 tokens/s, Avg generation throughput: 824.8 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.5%, Prefix cache hit rate: 22.8%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:58:44 [loggers.py:236] Engine 000: Avg prompt throughput: 785.7 tokens/s, Avg generation throughput: 765.4 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 21.8%
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59472 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45750 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59480 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41620 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45742 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55250 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:59494 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55248 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45766 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55232 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45790 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55264 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45784 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45820 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45816 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45770 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45806 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:41618 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45834 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45808 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:55578 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO: 127.0.0.1:45798 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:58:54 [loggers.py:236] Engine 000: Avg prompt throughput: 920.8 tokens/s, Avg generation throughput: 1024.1 tokens/s, Running: 14 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.8%, Prefix cache hit rate: 20.7%
+[0;36m(APIServer pid=15750)[0;0m INFO 12-10 10:58:59 [launcher.py:110] Shutting down FastAPI HTTP server.
diff --git a/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_throughput.json
new file mode 100644
index 0000000..0f814ab
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-a100/openai_gpt-oss-20b_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 156.32366928458214,
+ "num_requests": 1000,
+ "total_num_tokens": 738792,
+ "requests_per_second": 6.396983928131399,
+ "tokens_per_second": 4726.040550232053
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..bf144c3
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-39a9e102-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 197.48 \nTotal input tokens: 38358 \nTotal generated tokens: 40296 \nRequest throughput (req/s): 0.91 \nOutput token throughput (tok/s): 204.06 \nPeak output token throughput (tok/s): 372.00 \nPeak concurrent requests: 15.00 \nTotal Token throughput (tok/s): 398.30 \n---------------Time to First Token----------------\nMean TTFT (ms): 78.01 \nMedian TTFT (ms): 70.55 \nP99 TTFT (ms): 149.81 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 34.15 \nMedian TPOT (ms): 34.06 \nP99 TPOT (ms): 36.67 \n---------------Inter-token Latency----------------\nMean ITL (ms): 34.16 \nMedian ITL (ms): 33.55 \nP99 ITL (ms): 58.95 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..1b3d37f
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='RedHatAI/Qwen3-14B-FP8-dynamic', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-0e099965-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 205.53 \nTotal input tokens: 146694 \nTotal generated tokens: 155647 \nRequest throughput (req/s): 3.50 \nOutput token throughput (tok/s): 757.31 \nPeak output token throughput (tok/s): 1260.00 \nPeak concurrent requests: 59.00 \nTotal Token throughput (tok/s): 1471.06 \n---------------Time to First Token----------------\nMean TTFT (ms): 85.03 \nMedian TTFT (ms): 76.62 \nP99 TTFT (ms): 176.99 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 41.66 \nMedian TPOT (ms): 41.32 \nP99 TPOT (ms): 52.93 \n---------------Inter-token Latency----------------\nMean ITL (ms): 41.47 \nMedian ITL (ms): 39.19 \nP99 ITL (ms): 111.74 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
new file mode 100644
index 0000000..d2caafb
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_server.log
@@ -0,0 +1,1025 @@
+WARNING 12-10 19:52:40 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:52:40 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:52:40 [utils.py:253] non-default args: {'model_tag': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'host': '127.0.0.1', 'model': 'RedHatAI/Qwen3-14B-FP8-dynamic', 'trust_remote_code': True, 'max_model_len': 32768, 'max_num_seqs': 64}
+[0;36m(APIServer pid=4723)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:52:45 [model.py:637] Resolved architecture: Qwen3ForCausalLM
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:52:45 [model.py:1750] Using max model len 32768
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:52:45 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:52:49 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='RedHatAI/Qwen3-14B-FP8-dynamic', speculative_config=None, tokenizer='RedHatAI/Qwen3-14B-FP8-dynamic', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=32768, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=RedHatAI/Qwen3-14B-FP8-dynamic, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type': }, 'local_cache_dir': None}
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:52:50 [parallel_state.py:1200] world_size=1 rank=0 local_rank=0 distributed_init_method=tcp://172.17.0.3:34143 backend=nccl
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:52:50 [parallel_state.py:1408] rank 0 in world size 1 is assigned as DP rank 0, PP rank 0, PCP rank 0, TP rank 0, EP rank 0
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:52:50 [gpu_model_runner.py:3467] Starting to load model RedHatAI/Qwen3-14B-FP8-dynamic...
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:52:51 [cuda.py:411] Using FLASH_ATTN attention backend out of potential backends: ['FLASH_ATTN', 'FLASHINFER', 'TRITON_ATTN', 'FLEX_ATTENTION']
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Loading safetensors checkpoint shards: 0% Completed | 0/4 [00:00, ?it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Loading safetensors checkpoint shards: 25% Completed | 1/4 [00:00<00:00, 9.44it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Loading safetensors checkpoint shards: 50% Completed | 2/4 [00:00<00:00, 3.59it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Loading safetensors checkpoint shards: 75% Completed | 3/4 [00:00<00:00, 2.78it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 2.48it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Loading safetensors checkpoint shards: 100% Completed | 4/4 [00:01<00:00, 2.80it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:52:53 [default_loader.py:308] Loading weights took 1.53 seconds
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:52:53 [gpu_model_runner.py:3549] Model loading took 15.3388 GiB memory and 2.474270 seconds
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:01 [backends.py:655] Using cache directory: /root/.cache/vllm/torch_compile_cache/a39c13ae20/rank_0_0/backbone for vLLM's torch.compile
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:01 [backends.py:715] Dynamo bytecode transform time: 7.49 s
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:02 [backends.py:257] Cache the graph for dynamic shape for later use
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:07 [backends.py:288] Compiling a graph for dynamic shape takes 5.35 s
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:14 [monitor.py:34] torch.compile takes 12.84 s in total
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:15 [gpu_worker.py:359] Available KV cache memory: 12.31 GiB
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:16 [kv_cache_utils.py:1286] GPU KV cache size: 80,672 tokens
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:16 [kv_cache_utils.py:1291] Maximum concurrency for 32,768 tokens per request: 2.46x
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 0%| | 0/19 [00:00, ?it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 16%|█▌ | 3/19 [00:00<00:00, 21.09it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 32%|███▏ | 6/19 [00:00<00:00, 21.75it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 47%|████▋ | 9/19 [00:00<00:00, 22.12it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 63%|██████▎ | 12/19 [00:00<00:00, 22.46it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 79%|███████▉ | 15/19 [00:00<00:00, 22.86it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 95%|█████████▍| 18/19 [00:00<00:00, 23.35it/s]
Capturing CUDA graphs (mixed prefill-decode, PIECEWISE): 100%|██████████| 19/19 [00:00<00:00, 22.82it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m
Capturing CUDA graphs (decode, FULL): 0%| | 0/11 [00:00, ?it/s]
Capturing CUDA graphs (decode, FULL): 27%|██▋ | 3/11 [00:00<00:00, 21.12it/s]
Capturing CUDA graphs (decode, FULL): 55%|█████▍ | 6/11 [00:00<00:00, 22.80it/s]
Capturing CUDA graphs (decode, FULL): 82%|████████▏ | 9/11 [00:00<00:00, 23.92it/s]
Capturing CUDA graphs (decode, FULL): 100%|██████████| 11/11 [00:00<00:00, 23.81it/s]
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:18 [gpu_model_runner.py:4466] Graph capturing finished in 2 secs, took 0.25 GiB
+[0;36m(EngineCore_DP0 pid=4827)[0;0m INFO 12-10 19:53:18 [core.py:254] init engine (profile, create kv cache, warmup model) took 24.46 seconds
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:19 [api_server.py:1520] Supported tasks: ['generate']
+[0;36m(APIServer pid=4723)[0;0m WARNING 12-10 19:53:19 [model.py:1576] Default sampling parameters have been overridden by the model's Hugging Face generation config recommended from the model creator. If this is not intended, please relaunch vLLM instance with `--generation-config vllm`.
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:19 [serving_responses.py:194] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:19 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [serving_completion.py:73] Using default completion sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [serving_chat.py:133] Using default chat sampling params from model: {'temperature': 0.6, 'top_k': 20, 'top_p': 0.95}
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [api_server.py:1847] Starting vLLM API server 0 on http://127.0.0.1:8000
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:38] Available routes are:
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /openapi.json, Methods: GET, HEAD
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /docs, Methods: GET, HEAD
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /docs/oauth2-redirect, Methods: GET, HEAD
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /redoc, Methods: GET, HEAD
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /health, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /load, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /pause, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /resume, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /is_paused, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /tokenize, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /detokenize, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/models, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /version, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/responses, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/responses/{response_id}, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/responses/{response_id}/cancel, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/messages, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/chat/completions, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/completions, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/audio/transcriptions, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/audio/translations, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /scale_elastic_ep, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /is_scaling_elastic_ep, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /inference/v1/generate, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /ping, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /ping, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /invocations, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /metrics, Methods: GET
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /classify, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/embeddings, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /score, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/score, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /rerank, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v1/rerank, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /v2/rerank, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:20 [launcher.py:46] Route: /pooling, Methods: POST
+[0;36m(APIServer pid=4723)[0;0m INFO: Started server process [4723]
+[0;36m(APIServer pid=4723)[0;0m INFO: Waiting for application startup.
+[0;36m(APIServer pid=4723)[0;0m INFO: Application startup complete.
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:46822 - "GET /v1/models HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:40 [loggers.py:236] Engine 000: Avg prompt throughput: 8.3 tokens/s, Avg generation throughput: 30.0 tokens/s, Running: 4 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.3%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:53:50 [loggers.py:236] Engine 000: Avg prompt throughput: 170.5 tokens/s, Avg generation throughput: 125.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.8%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:54:00 [loggers.py:236] Engine 000: Avg prompt throughput: 249.4 tokens/s, Avg generation throughput: 201.4 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:54:10 [loggers.py:236] Engine 000: Avg prompt throughput: 214.4 tokens/s, Avg generation throughput: 186.6 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:54:20 [loggers.py:236] Engine 000: Avg prompt throughput: 217.3 tokens/s, Avg generation throughput: 185.6 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:54:30 [loggers.py:236] Engine 000: Avg prompt throughput: 379.7 tokens/s, Avg generation throughput: 219.0 tokens/s, Running: 11 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51534 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32772 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:54:40 [loggers.py:236] Engine 000: Avg prompt throughput: 400.9 tokens/s, Avg generation throughput: 219.2 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32788 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:54:50 [loggers.py:236] Engine 000: Avg prompt throughput: 215.3 tokens/s, Avg generation throughput: 201.3 tokens/s, Running: 5 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:32776 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:55:00 [loggers.py:236] Engine 000: Avg prompt throughput: 298.4 tokens/s, Avg generation throughput: 234.4 tokens/s, Running: 10 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57036 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:55:10 [loggers.py:236] Engine 000: Avg prompt throughput: 301.7 tokens/s, Avg generation throughput: 316.4 tokens/s, Running: 13 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.4%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51564 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57028 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57060 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:55:20 [loggers.py:236] Engine 000: Avg prompt throughput: 160.7 tokens/s, Avg generation throughput: 304.3 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40602 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:55:30 [loggers.py:236] Engine 000: Avg prompt throughput: 36.4 tokens/s, Avg generation throughput: 211.4 tokens/s, Running: 6 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.5%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:55:40 [loggers.py:236] Engine 000: Avg prompt throughput: 278.6 tokens/s, Avg generation throughput: 206.7 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 5.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:55:50 [loggers.py:236] Engine 000: Avg prompt throughput: 172.8 tokens/s, Avg generation throughput: 236.4 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51886 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57016 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:57052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51550 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:56:00 [loggers.py:236] Engine 000: Avg prompt throughput: 157.6 tokens/s, Avg generation throughput: 255.1 tokens/s, Running: 8 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36576 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:56:10 [loggers.py:236] Engine 000: Avg prompt throughput: 65.8 tokens/s, Avg generation throughput: 130.1 tokens/s, Running: 1 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:40598 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49190 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49154 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:56:20 [loggers.py:236] Engine 000: Avg prompt throughput: 180.0 tokens/s, Avg generation throughput: 182.5 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 3.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51880 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49144 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36574 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49176 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36600 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36610 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:56:30 [loggers.py:236] Engine 000: Avg prompt throughput: 189.6 tokens/s, Avg generation throughput: 187.0 tokens/s, Running: 9 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.2%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36614 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49906 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:49164 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:60990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:56:40 [loggers.py:236] Engine 000: Avg prompt throughput: 139.6 tokens/s, Avg generation throughput: 259.9 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 4.1%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:56:50 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 131.6 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 1.6%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:57:00 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 17.3 tokens/s, Running: 0 reqs, Waiting: 0 reqs, GPU KV cache usage: 0.0%, Prefix cache hit rate: 0.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:57:10 [loggers.py:236] Engine 000: Avg prompt throughput: 323.0 tokens/s, Avg generation throughput: 108.9 tokens/s, Running: 12 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.7%, Prefix cache hit rate: 7.4%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:57:20 [loggers.py:236] Engine 000: Avg prompt throughput: 1081.4 tokens/s, Avg generation throughput: 523.3 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.6%, Prefix cache hit rate: 25.8%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:57:30 [loggers.py:236] Engine 000: Avg prompt throughput: 1146.2 tokens/s, Avg generation throughput: 806.0 tokens/s, Running: 34 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.4%, Prefix cache hit rate: 38.4%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:57:40 [loggers.py:236] Engine 000: Avg prompt throughput: 610.1 tokens/s, Avg generation throughput: 902.5 tokens/s, Running: 42 reqs, Waiting: 0 reqs, GPU KV cache usage: 17.6%, Prefix cache hit rate: 43.4%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47930 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:57:50 [loggers.py:236] Engine 000: Avg prompt throughput: 625.7 tokens/s, Avg generation throughput: 866.7 tokens/s, Running: 31 reqs, Waiting: 0 reqs, GPU KV cache usage: 15.5%, Prefix cache hit rate: 47.7%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:58:00 [loggers.py:236] Engine 000: Avg prompt throughput: 936.5 tokens/s, Avg generation throughput: 828.3 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.6%, Prefix cache hit rate: 43.0%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59128 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:58:10 [loggers.py:236] Engine 000: Avg prompt throughput: 997.0 tokens/s, Avg generation throughput: 957.7 tokens/s, Running: 41 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.7%, Prefix cache hit rate: 38.5%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47888 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47876 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:58:20 [loggers.py:236] Engine 000: Avg prompt throughput: 537.1 tokens/s, Avg generation throughput: 829.6 tokens/s, Running: 36 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.6%, Prefix cache hit rate: 36.5%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50280 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:58:30 [loggers.py:236] Engine 000: Avg prompt throughput: 978.6 tokens/s, Avg generation throughput: 953.1 tokens/s, Running: 53 reqs, Waiting: 0 reqs, GPU KV cache usage: 19.2%, Prefix cache hit rate: 33.2%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50526 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51094 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48074 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48006 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51020 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:58:40 [loggers.py:236] Engine 000: Avg prompt throughput: 528.9 tokens/s, Avg generation throughput: 1091.1 tokens/s, Running: 39 reqs, Waiting: 0 reqs, GPU KV cache usage: 18.0%, Prefix cache hit rate: 31.7%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59078 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50270 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51004 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:58:50 [loggers.py:236] Engine 000: Avg prompt throughput: 963.7 tokens/s, Avg generation throughput: 833.9 tokens/s, Running: 31 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.2%, Prefix cache hit rate: 29.3%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50292 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47958 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59086 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48090 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51026 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:59:00 [loggers.py:236] Engine 000: Avg prompt throughput: 877.4 tokens/s, Avg generation throughput: 721.8 tokens/s, Running: 28 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.8%, Prefix cache hit rate: 27.4%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47900 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:59:10 [loggers.py:236] Engine 000: Avg prompt throughput: 506.8 tokens/s, Avg generation throughput: 718.1 tokens/s, Running: 25 reqs, Waiting: 0 reqs, GPU KV cache usage: 10.1%, Prefix cache hit rate: 26.4%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47944 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:59:20 [loggers.py:236] Engine 000: Avg prompt throughput: 898.4 tokens/s, Avg generation throughput: 827.2 tokens/s, Running: 36 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.4%, Prefix cache hit rate: 24.9%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50510 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47904 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:59:30 [loggers.py:236] Engine 000: Avg prompt throughput: 784.6 tokens/s, Avg generation throughput: 736.9 tokens/s, Running: 26 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.3%, Prefix cache hit rate: 23.7%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50286 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:59:40 [loggers.py:236] Engine 000: Avg prompt throughput: 786.4 tokens/s, Avg generation throughput: 757.8 tokens/s, Running: 33 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.5%, Prefix cache hit rate: 22.5%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50528 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50482 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 19:59:50 [loggers.py:236] Engine 000: Avg prompt throughput: 556.2 tokens/s, Avg generation throughput: 802.9 tokens/s, Running: 32 reqs, Waiting: 0 reqs, GPU KV cache usage: 13.2%, Prefix cache hit rate: 21.8%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51068 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50522 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48038 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53502 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47970 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50254 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50246 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51124 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:36416 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51122 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51082 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59132 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50464 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59112 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53516 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59096 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 20:00:00 [loggers.py:236] Engine 000: Avg prompt throughput: 1196.3 tokens/s, Avg generation throughput: 791.3 tokens/s, Running: 34 reqs, Waiting: 0 reqs, GPU KV cache usage: 14.5%, Prefix cache hit rate: 20.4%
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47854 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47862 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50476 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47938 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50474 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51110 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48052 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:54118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51136 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50490 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50504 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:59118 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:50518 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53916 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53928 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53940 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53956 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:48050 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53914 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47990 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53964 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53966 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53974 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:47978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53978 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:53982 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO: 127.0.0.1:51064 - "POST /v1/completions HTTP/1.1" 200 OK
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 20:00:10 [loggers.py:236] Engine 000: Avg prompt throughput: 335.4 tokens/s, Avg generation throughput: 944.3 tokens/s, Running: 27 reqs, Waiting: 0 reqs, GPU KV cache usage: 11.5%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 20:00:20 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 432.7 tokens/s, Running: 7 reqs, Waiting: 0 reqs, GPU KV cache usage: 6.7%, Prefix cache hit rate: 20.0%
+[0;36m(APIServer pid=4723)[0;0m INFO 12-10 20:00:30 [loggers.py:236] Engine 000: Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 138.2 tokens/s, Running: 2 reqs, Waiting: 0 reqs, GPU KV cache usage: 2.5%, Prefix cache hit rate: 20.0%
diff --git a/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
new file mode 100644
index 0000000..a1bf191
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-ada5000/RedHatAI_Qwen3-14B-FP8-dynamic_tp1_throughput.json
@@ -0,0 +1,7 @@
+{
+ "elapsed_time": 405.9800275520247,
+ "num_requests": 1000,
+ "total_num_tokens": 741334,
+ "requests_per_second": 2.463175358723414,
+ "tokens_per_second": 1826.0356413838635
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json b/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
new file mode 100644
index 0000000..be0cfe4
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps1.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=180, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=1.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-341eb0ef-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 1.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 180 \nFailed requests: 0 \nRequest rate configured (RPS): 1.00 \nBenchmark duration (s): 184.76 \nTotal input tokens: 38358 \nTotal generated tokens: 40087 \nRequest throughput (req/s): 0.97 \nOutput token throughput (tok/s): 216.97 \nPeak output token throughput (tok/s): 453.00 \nPeak concurrent requests: 9.00 \nTotal Token throughput (tok/s): 424.58 \n---------------Time to First Token----------------\nMean TTFT (ms): 40.37 \nMedian TTFT (ms): 37.70 \nP99 TTFT (ms): 70.13 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 11.50 \nMedian TPOT (ms): 11.56 \nP99 TPOT (ms): 16.13 \n---------------Inter-token Latency----------------\nMean ITL (ms): 11.24 \nMedian ITL (ms): 11.23 \nP99 ITL (ms): 19.54 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json b/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
new file mode 100644
index 0000000..be18451
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_qps4.0_latency.json
@@ -0,0 +1,4 @@
+{
+ "success": true,
+ "raw_output": "Namespace(subparser='bench', bench_type='serve', dispatch_function=, seed=0, num_prompts=720, dataset_name='sharegpt', no_stream=False, dataset_path='ShareGPT_V3_unfiltered_cleaned_split.json', no_oversample=False, skip_chat_template=False, disable_shuffle=False, custom_output_len=256, spec_bench_output_len=256, spec_bench_category=None, sonnet_input_len=550, sonnet_output_len=150, sonnet_prefix_len=200, sharegpt_output_len=None, blazedit_min_distance=0.0, blazedit_max_distance=1.0, random_input_len=1024, random_output_len=128, random_range_ratio=0.0, random_prefix_len=0, random_batch_size=1, no_reranker=False, random_mm_base_items_per_request=1, random_mm_num_mm_items_range_ratio=0.0, random_mm_limit_mm_per_prompt={'image': 255, 'video': 1}, random_mm_bucket_config={(256, 256, 1): 0.5, (720, 1280, 1): 0.5, (720, 1280, 16): 0.0}, hf_subset=None, hf_split=None, hf_name=None, hf_output_len=None, prefix_repetition_prefix_len=256, prefix_repetition_suffix_len=256, prefix_repetition_num_prefixes=10, prefix_repetition_output_len=128, label=None, backend='openai', base_url='http://127.0.0.1:8000', host='127.0.0.1', port=8000, endpoint='/v1/completions', header=None, max_concurrency=None, model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', tokenizer=None, use_beam_search=False, logprobs=None, request_rate=4.0, burstiness=1.0, trust_remote_code=True, disable_tqdm=False, num_warmups=0, profile=False, save_result=False, save_detailed=False, append_result=False, metadata=None, result_dir=None, result_filename=None, ignore_eos=False, percentile_metrics=None, metric_percentiles='99', goodput=None, request_id_prefix='bench-34321bd4-', top_p=None, top_k=None, min_p=None, temperature=None, frequency_penalty=None, presence_penalty=None, repetition_penalty=None, tokenizer_mode='auto', served_model_name=None, lora_modules=None, ramp_up_strategy=None, ramp_up_start_rps=None, ramp_up_end_rps=None, ready_check_timeout_sec=600, extra_body=None)\nStarting initial single prompt test run...\nWaiting for endpoint to become up in 600 seconds\nInitial test run completed.\nStarting main benchmark run...\nTraffic request rate: 4.0\nBurstiness factor: 1.0 (Poisson process)\nMaximum request concurrency: None\ntip: install termplotlib and gnuplot to plot the metrics\n============ Serving Benchmark Result ============\nSuccessful requests: 720 \nFailed requests: 0 \nRequest rate configured (RPS): 4.00 \nBenchmark duration (s): 192.26 \nTotal input tokens: 146694 \nTotal generated tokens: 153536 \nRequest throughput (req/s): 3.74 \nOutput token throughput (tok/s): 798.60 \nPeak output token throughput (tok/s): 1194.00 \nPeak concurrent requests: 46.00 \nTotal Token throughput (tok/s): 1561.61 \n---------------Time to First Token----------------\nMean TTFT (ms): 54.49 \nMedian TTFT (ms): 52.17 \nP99 TTFT (ms): 97.75 \n-----Time per Output Token (excl. 1st token)------\nMean TPOT (ms): 25.72 \nMedian TPOT (ms): 25.55 \nP99 TPOT (ms): 31.80 \n---------------Inter-token Latency----------------\nMean ITL (ms): 25.55 \nMedian ITL (ms): 24.86 \nP99 ITL (ms): 53.01 \n==================================================\n"
+}
\ No newline at end of file
diff --git a/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log b/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
new file mode 100644
index 0000000..54d348e
--- /dev/null
+++ b/benchmarks/benchmark_results_nvidia-ada5000/cpatonn_Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit_tp1_server.log
@@ -0,0 +1,1025 @@
+WARNING 12-10 20:10:10 [argparse_utils.py:195] With `vllm serve`, you should provide the model as a positional argument or in a config file instead of via the `--model` option. The `--model` option will be removed in v0.13.
+[0;36m(APIServer pid=5918)[0;0m INFO 12-10 20:10:10 [api_server.py:1772] vLLM API server version 0.12.0
+[0;36m(APIServer pid=5918)[0;0m INFO 12-10 20:10:10 [utils.py:253] non-default args: {'model_tag': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'host': '127.0.0.1', 'model': 'cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', 'trust_remote_code': True, 'max_model_len': 24576, 'max_num_seqs': 64}
+[0;36m(APIServer pid=5918)[0;0m The argument `trust_remote_code` is to be used with Auto classes. It has no effect here and is ignored.
+[0;36m(APIServer pid=5918)[0;0m INFO 12-10 20:10:17 [model.py:637] Resolved architecture: Qwen3MoeForCausalLM
+[0;36m(APIServer pid=5918)[0;0m INFO 12-10 20:10:17 [model.py:1750] Using max model len 24576
+[0;36m(APIServer pid=5918)[0;0m INFO 12-10 20:10:18 [scheduler.py:228] Chunked prefill is enabled with max_num_batched_tokens=2048.
+[0;36m(EngineCore_DP0 pid=6022)[0;0m INFO 12-10 20:10:23 [core.py:93] Initializing a V1 LLM engine (v0.12.0) with config: model='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', speculative_config=None, tokenizer='cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.bfloat16, max_seq_len=24576, download_dir=None, load_format=auto, tensor_parallel_size=1, pipeline_parallel_size=1, data_parallel_size=1, disable_custom_all_reduce=False, quantization=compressed-tensors, enforce_eager=False, kv_cache_dtype=auto, device_config=cuda, structured_outputs_config=StructuredOutputsConfig(backend='auto', disable_fallback=False, disable_any_whitespace=False, disable_additional_properties=False, reasoning_parser='', reasoning_parser_plugin='', enable_in_reasoning=False), observability_config=ObservabilityConfig(show_hidden_metrics_for_version=None, otlp_traces_endpoint=None, collect_detailed_traces=None, kv_cache_metrics=False, kv_cache_metrics_sample=0.01), seed=0, served_model_name=cpatonn/Qwen3-Coder-30B-A3B-Instruct-GPTQ-4bit, enable_prefix_caching=True, enable_chunked_prefill=True, pooler_config=None, compilation_config={'level': None, 'mode': , 'debug_dump_path': None, 'cache_dir': '', 'compile_cache_save_format': 'binary', 'backend': 'inductor', 'custom_ops': ['none'], 'splitting_ops': ['vllm::unified_attention', 'vllm::unified_attention_with_output', 'vllm::unified_mla_attention', 'vllm::unified_mla_attention_with_output', 'vllm::mamba_mixer2', 'vllm::mamba_mixer', 'vllm::short_conv', 'vllm::linear_attention', 'vllm::plamo2_mamba_mixer', 'vllm::gdn_attention_core', 'vllm::kda_attention', 'vllm::sparse_attn_indexer'], 'compile_mm_encoder': False, 'compile_sizes': [], 'inductor_compile_config': {'enable_auto_functionalized_v2': False, 'combo_kernels': True, 'benchmark_combo_kernel': True}, 'inductor_passes': {}, 'cudagraph_mode': , 'cudagraph_num_of_warmups': 1, 'cudagraph_capture_sizes': [1, 2, 4, 8, 16, 24, 32, 40, 48, 56, 64, 72, 80, 88, 96, 104, 112, 120, 128], 'cudagraph_copy_inputs': False, 'cudagraph_specialize_lora': True, 'use_inductor_graph_partition': False, 'pass_config': {'fuse_norm_quant': False, 'fuse_act_quant': False, 'fuse_attn_quant': False, 'eliminate_noops': True, 'enable_sp': False, 'fuse_gemm_comms': False, 'fuse_allreduce_rms': False}, 'max_cudagraph_capture_size': 128, 'dynamic_shapes_config': {'type':