forked from flagos-ai/vllm-plugin-FL
-
Notifications
You must be signed in to change notification settings - Fork 3
Expand file tree
/
Copy pathrun_benchmark.sh
More file actions
107 lines (89 loc) · 3.52 KB
/
Copy pathrun_benchmark.sh
File metadata and controls
107 lines (89 loc) · 3.52 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
#!/bin/bash
# Copyright 2026 FlagOS Contributors
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
set -e
# Arguments
MODEL_PATH=${1:?"Please provide model path, e.g.: ./run_benchmark.sh /data/models/Qwen/Qwen3-4B/"}
# Output directory
OUTPUT_DIR=bench_results
mkdir -p "${OUTPUT_DIR}"
echo "=== Starting benchmark for model: ${MODEL_PATH} ==="
echo "Results will be saved to: ${OUTPUT_DIR}/"
echo ""
# ============================================================================
# Configuration based on mode
# ============================================================================
declare -A THROUGHPUT_SCENARIOS=(
# input_len output_len num_prompts
["chat_1k"]="1024 1024 300"
["chat_4k"]="4096 1024 300"
["chat_6k"]="6144 1024 300"
)
declare -A LATENCY_SCENARIOS=(
# input_len output_len batch_size num_iters
["batch_8"]="4096 1024 8 10"
)
# ============================================================================
# Throughput Tests
# ============================================================================
echo "==================== THROUGHPUT TESTS ===================="
for scenario in "${!THROUGHPUT_SCENARIOS[@]}"; do
read input_len output_len num_prompts <<< "${THROUGHPUT_SCENARIOS[$scenario]}"
output_file="${OUTPUT_DIR}/throughput_${scenario}.json"
echo ""
echo "--- Throughput: ${scenario} (input=${input_len}, output=${output_len}, prompts=${num_prompts}) ---"
vllm bench throughput \
--model "${MODEL_PATH}" \
--input-len "${input_len}" \
--output-len "${output_len}" \
--num-prompts "${num_prompts}" \
--trust-remote-code \
--dtype auto \
--enforce-eager \
--output-json "${output_file}"
echo "Saved: ${output_file}"
done
# ============================================================================
# Latency Tests
# ============================================================================
echo ""
echo "==================== LATENCY TESTS ===================="
for scenario in "${!LATENCY_SCENARIOS[@]}"; do
read input_len output_len batch_size num_iters <<< "${LATENCY_SCENARIOS[$scenario]}"
output_file="${OUTPUT_DIR}/latency_${scenario}.json"
echo ""
echo "--- Latency: ${scenario} (input=${input_len}, output=${output_len}, batch=${batch_size}, iters=${num_iters}) ---"
vllm bench latency \
--model "${MODEL_PATH}" \
--input-len "${input_len}" \
--output-len "${output_len}" \
--batch-size "${batch_size}" \
--num-iters "${num_iters}" \
--trust-remote-code \
--dtype auto \
--enforce-eager \
--output-json "${output_file}"
echo "Saved: ${output_file}"
done
echo ""
echo "==================== BENCHMARK COMPLETED ===================="
echo "All results saved to: ${OUTPUT_DIR}/"
echo ""
echo "Files generated:"
ls -la "${OUTPUT_DIR}"/*.json
# Collect and summarize results
echo ""
echo "=== Collecting benchmark results... ==="
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
python3 "${SCRIPT_DIR}/collect_benchmark_results.py" "${OUTPUT_DIR}"