test/scripts/docker/entrypoint.sh

616 lines
18 KiB
Bash

#!/usr/bin/env bash
set -euo pipefail
DEFAULT_CMD_1="${APP_PYTHON_BIN:-/opt/venv/bin/python}"
DEFAULT_CMD_2="start.py"
ACTIVE_ACCELERATOR=""
normalize_topology() {
local raw="${ASR_DEPLOY_TOPOLOGY:-isolated}"
raw="${raw,,}"
case "$raw" in
isolated|sharded|auto)
echo "$raw"
return 0
;;
*)
echo "[entrypoint] Invalid ASR_DEPLOY_TOPOLOGY=${raw}, fallback to isolated" >&2
echo "isolated"
return 0
;;
esac
}
detect_accelerator() {
local configured="${ACCELERATOR:-auto}"
configured="${configured,,}"
case "$configured" in
nvidia|metax|iluvatar|mthreads|cpu)
echo "$configured"
return 0
;;
cuda)
echo "nvidia"
return 0
;;
maca|muxi|mx)
echo "metax"
return 0
;;
ix|tianshu|天数)
echo "iluvatar"
return 0
;;
mthreads|musa|moorethreads|摩尔线程)
echo "mthreads"
return 0
;;
esac
if command -v mx-smi >/dev/null 2>&1; then
echo "metax"
return 0
fi
if command -v ixsmi >/dev/null 2>&1; then
echo "iluvatar"
return 0
fi
if command -v mthreads-gmi >/dev/null 2>&1; then
echo "mthreads"
return 0
fi
if command -v nvidia-smi >/dev/null 2>&1; then
echo "nvidia"
return 0
fi
echo "cpu"
}
list_nvidia_devices() {
if command -v nvidia-smi >/dev/null 2>&1; then
nvidia-smi --query-gpu=index --format=csv,noheader,nounits | sed '/^$/d'
fi
}
list_metax_devices() {
if ! command -v mx-smi >/dev/null 2>&1; then
return 0
fi
local output
output="$(mx-smi -L 2>/dev/null || true)"
if [[ -n "$output" ]]; then
echo "$output" | awk 'NF { print NR - 1 }'
return 0
fi
output="$(mx-smi 2>/dev/null || true)"
if [[ -n "$output" ]]; then
echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }'
fi
}
list_iluvatar_devices() {
if ! command -v ixsmi >/dev/null 2>&1; then
return 0
fi
local output
output="$(ixsmi -L 2>/dev/null || true)"
if [[ -n "$output" ]]; then
echo "$output" | awk 'NF { print NR - 1 }'
return 0
fi
output="$(ixsmi 2>/dev/null || true)"
if [[ -n "$output" ]]; then
echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }'
fi
}
list_mthreads_devices() {
if ! command -v mthreads-gmi >/dev/null 2>&1; then
return 0
fi
local output
output="$(mthreads-gmi -L 2>/dev/null || true)"
if [[ -n "$output" ]]; then
echo "$output" | awk 'NF { print NR - 1 }'
return 0
fi
output="$(mthreads-gmi list 2>/dev/null || true)"
if [[ -n "$output" ]]; then
echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }'
return 0
fi
output="$(mthreads-gmi 2>/dev/null || true)"
if [[ -n "$output" ]]; then
echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }'
fi
}
normalize_visible_devices() {
local accelerator="$1"
local shared="${ASR_VISIBLE_DEVICES:-}"
shared="${shared// /}"
if [[ -n "$shared" && "$shared" != "none" && "$shared" != "void" ]]; then
if [[ "$shared" == "all" ]]; then
case "$accelerator" in
nvidia) mapfile -t gpu_indexes < <(list_nvidia_devices) ;;
metax) mapfile -t gpu_indexes < <(list_metax_devices) ;;
iluvatar) mapfile -t gpu_indexes < <(list_iluvatar_devices) ;;
mthreads) mapfile -t gpu_indexes < <(list_mthreads_devices) ;;
*) gpu_indexes=() ;;
esac
if [[ ${#gpu_indexes[@]} -eq 0 ]]; then
echo ""
return 0
fi
local shared_joined
shared_joined=$(IFS=,; echo "${gpu_indexes[*]}")
echo "$shared_joined"
return 0
fi
echo "$shared"
return 0
fi
local raw=""
case "$accelerator" in
nvidia)
raw="${CUDA_VISIBLE_DEVICES:-}"
;;
metax)
raw="${METAX_VISIBLE_DEVICES:-${MACA_VISIBLE_DEVICES:-${MX_VISIBLE_DEVICES:-}}}"
;;
iluvatar)
raw="${ILUVATAR_VISIBLE_DEVICES:-${IX_VISIBLE_DEVICES:-${CUDA_VISIBLE_DEVICES:-}}}"
;;
mthreads)
raw="${MTHREADS_VISIBLE_DEVICES:-${MUSA_VISIBLE_DEVICES:-${CUDA_VISIBLE_DEVICES:-}}}"
;;
*)
echo ""
return 0
;;
esac
raw="${raw// /}"
if [[ -z "$raw" || "$raw" == "none" || "$raw" == "void" ]]; then
echo ""
return 0
fi
if [[ "$raw" == "all" ]]; then
case "$accelerator" in
nvidia) mapfile -t gpu_indexes < <(list_nvidia_devices) ;;
metax) mapfile -t gpu_indexes < <(list_metax_devices) ;;
iluvatar) mapfile -t gpu_indexes < <(list_iluvatar_devices) ;;
mthreads) mapfile -t gpu_indexes < <(list_mthreads_devices) ;;
esac
if [[ ${#gpu_indexes[@]} -eq 0 ]]; then
echo ""
return 0
fi
local joined
joined=$(IFS=,; echo "${gpu_indexes[*]}")
echo "$joined"
return 0
fi
echo "$raw"
}
export_visible_devices_aliases() {
local accelerator="$1"
local devices_csv="$2"
if [[ -z "$devices_csv" ]]; then
return 0
fi
export ASR_VISIBLE_DEVICES="$devices_csv"
case "$accelerator" in
nvidia)
export CUDA_VISIBLE_DEVICES="$devices_csv"
;;
metax)
export METAX_VISIBLE_DEVICES="$devices_csv"
export MACA_VISIBLE_DEVICES="$devices_csv"
export MX_VISIBLE_DEVICES="$devices_csv"
;;
iluvatar)
export ILUVATAR_VISIBLE_DEVICES="$devices_csv"
export IX_VISIBLE_DEVICES="$devices_csv"
export CUDA_VISIBLE_DEVICES="$devices_csv"
;;
mthreads)
export MTHREADS_VISIBLE_DEVICES="$devices_csv"
export MUSA_VISIBLE_DEVICES="$devices_csv"
export CUDA_VISIBLE_DEVICES="$devices_csv"
;;
esac
}
supports_sharded_topology() {
local accelerator="$1"
case "$accelerator" in
nvidia|metax|iluvatar|mthreads)
return 0
;;
*)
return 1
;;
esac
}
start_backend_process() {
local accelerator="$1"
local device="$2"
local port="$3"
local log_file="$4"
local bind_host="$5"
local workers="${WORKERS:-1}"
if [[ "$accelerator" == "nvidia" && -n "$device" ]]; then
CUDA_VISIBLE_DEVICES="$device" \
ACCELERATOR="nvidia" \
DEVICE="cuda:0" \
WORKERS="$workers" \
HOST="$bind_host" \
PORT="$port" \
LOG_FILE="$log_file" \
"$DEFAULT_CMD_1" "$DEFAULT_CMD_2" &
return 0
fi
if [[ "$accelerator" == "metax" && -n "$device" ]]; then
METAX_VISIBLE_DEVICES="$device" \
MACA_VISIBLE_DEVICES="$device" \
MX_VISIBLE_DEVICES="$device" \
ACCELERATOR="metax" \
DEVICE="cuda:0" \
WORKERS="$workers" \
HOST="$bind_host" \
PORT="$port" \
LOG_FILE="$log_file" \
"$DEFAULT_CMD_1" "$DEFAULT_CMD_2" &
return 0
fi
if [[ "$accelerator" == "iluvatar" && -n "$device" ]]; then
ILUVATAR_VISIBLE_DEVICES="$device" \
IX_VISIBLE_DEVICES="$device" \
CUDA_VISIBLE_DEVICES="$device" \
ACCELERATOR="iluvatar" \
DEVICE="cuda:0" \
WORKERS="$workers" \
HOST="$bind_host" \
PORT="$port" \
LOG_FILE="$log_file" \
"$DEFAULT_CMD_1" "$DEFAULT_CMD_2" &
return 0
fi
if [[ "$accelerator" == "mthreads" && -n "$device" ]]; then
MTHREADS_VISIBLE_DEVICES="$device" \
MUSA_VISIBLE_DEVICES="$device" \
CUDA_VISIBLE_DEVICES="$device" \
ACCELERATOR="mthreads" \
DEVICE="cuda:0" \
WORKERS="$workers" \
HOST="$bind_host" \
PORT="$port" \
LOG_FILE="$log_file" \
"$DEFAULT_CMD_1" "$DEFAULT_CMD_2" &
return 0
fi
ACCELERATOR="$accelerator" \
WORKERS="$workers" \
HOST="$bind_host" \
PORT="$port" \
LOG_FILE="$log_file" \
"$DEFAULT_CMD_1" "$DEFAULT_CMD_2" &
}
exec_backend_process() {
local accelerator="$1"
local device="$2"
local port="$3"
local log_file="$4"
local bind_host="$5"
local workers="${WORKERS:-1}"
if [[ "$accelerator" == "nvidia" && -n "$device" ]]; then
export CUDA_VISIBLE_DEVICES="$device"
export DEVICE="cuda:0"
elif [[ "$accelerator" == "metax" && -n "$device" ]]; then
export METAX_VISIBLE_DEVICES="$device"
export MACA_VISIBLE_DEVICES="$device"
export MX_VISIBLE_DEVICES="$device"
export DEVICE="cuda:0"
elif [[ "$accelerator" == "iluvatar" && -n "$device" ]]; then
export ILUVATAR_VISIBLE_DEVICES="$device"
export IX_VISIBLE_DEVICES="$device"
export CUDA_VISIBLE_DEVICES="$device"
export DEVICE="cuda:0"
elif [[ "$accelerator" == "mthreads" && -n "$device" ]]; then
export MTHREADS_VISIBLE_DEVICES="$device"
export MUSA_VISIBLE_DEVICES="$device"
export CUDA_VISIBLE_DEVICES="$device"
export DEVICE="cuda:0"
fi
export ACCELERATOR="$accelerator"
export WORKERS="$workers"
export HOST="$bind_host"
export PORT="$port"
export LOG_FILE="$log_file"
exec "$DEFAULT_CMD_1" "$DEFAULT_CMD_2"
}
start_sharded_backend_mode() {
local accelerator="$1"
local devices_csv="$2"
local bind_host="${HOST:-0.0.0.0}"
local public_port="${PORT:-8000}"
local log_file="${LOG_FILE:-/app/data/logs/qwen3-asr.log}"
local workers="${WORKERS:-1}"
local shard_size=0
if [[ -n "$devices_csv" ]]; then
IFS=',' read -r -a devs <<< "$devices_csv"
shard_size="${#devs[@]}"
fi
if ! supports_sharded_topology "$accelerator"; then
echo "[entrypoint] accelerator=${accelerator} does not support sharded topology, fallback to isolated"
return 1
fi
if [[ -z "$devices_csv" ]]; then
echo "[entrypoint] sharded topology requires visible GPU devices, fallback to isolated"
return 1
fi
if ! [[ "$shard_size" =~ ^[0-9]+$ ]] || (( shard_size < 2 )); then
echo "[entrypoint] sharded topology requires at least 2 devices, got shard_size=${shard_size}, fallback to isolated"
return 1
fi
echo "[entrypoint] Starting sharded backend: accelerator=${accelerator} devices=${devices_csv} shard_size=${shard_size} bind=${bind_host}:${public_port}"
export ASR_ACTIVE_TOPOLOGY="sharded"
export_visible_devices_aliases "$accelerator" "$devices_csv"
exec_backend_process "$accelerator" "" "$public_port" "$log_file" "$bind_host"
}
wait_for_port() {
local host="$1"
local port="$2"
local timeout_sec="$3"
local deadline=$((SECONDS + timeout_sec))
while (( SECONDS < deadline )); do
if (echo >/dev/tcp/"$host"/"$port") >/dev/null 2>&1; then
return 0
fi
sleep 1
done
return 1
}
is_positive_integer() {
[[ "$1" =~ ^[0-9]+$ ]] && (( "$1" > 0 ))
}
start_internal_nginx_mode() {
local devices_csv="$1"
local accelerator="${ACTIVE_ACCELERATOR:-auto}"
local bind_host="${MULTI_GPU_BIND_HOST:-127.0.0.1}"
local base_port="18000"
local public_port="${PORT:-8000}"
local ready_timeout="${MULTI_GPU_READY_TIMEOUT:-180}"
local rate_limit_rps="${NGINX_RATE_LIMIT_RPS:-0}"
local rate_limit_burst="${NGINX_RATE_LIMIT_BURST:-0}"
local valid_devices=()
local devices=()
if [[ -n "$devices_csv" ]]; then
IFS=',' read -r -a devices <<< "$devices_csv"
for dev in "${devices[@]}"; do
if [[ -n "$dev" ]]; then
valid_devices+=("$dev")
fi
done
fi
if [[ "$rate_limit_rps" != "0" ]] && ! is_positive_integer "$rate_limit_rps"; then
echo "[entrypoint] Invalid NGINX_RATE_LIMIT_RPS=${rate_limit_rps}, fallback to 0 (disabled)"
rate_limit_rps="0"
fi
if [[ "$rate_limit_burst" != "0" ]] && ! is_positive_integer "$rate_limit_burst"; then
echo "[entrypoint] Invalid NGINX_RATE_LIMIT_BURST=${rate_limit_burst}, fallback to 0"
rate_limit_burst="0"
fi
if [[ "$rate_limit_rps" != "0" && "$rate_limit_burst" == "0" ]]; then
# By default, give short burst headroom equal to rate limit.
rate_limit_burst="$rate_limit_rps"
fi
if ! command -v nginx >/dev/null 2>&1; then
if [[ ${#valid_devices[@]} -le 1 ]]; then
local direct_device=""
local direct_log_file="${LOG_FILE:-/app/data/logs/qwen3-asr.log}"
local direct_bind_host="${HOST:-0.0.0.0}"
if [[ ${#valid_devices[@]} -eq 1 ]]; then
direct_device="${valid_devices[0]}"
fi
echo "[entrypoint] nginx not found; starting single backend directly on ${direct_bind_host}:${public_port}"
exec_backend_process "$accelerator" "$direct_device" "$public_port" "$direct_log_file" "$direct_bind_host"
fi
echo "[entrypoint] nginx not found in image; cannot start multi-backend proxy mode"
exit 1
fi
local backend_ports=()
local backend_pids=()
local idx=0
local default_log_file="${LOG_FILE:-/app/data/logs/qwen3-asr.log}"
if [[ ${#valid_devices[@]} -eq 0 ]]; then
local single_port="$base_port"
local single_log_file="${default_log_file%.log}-gpu0.log"
backend_ports+=("$single_port")
echo "[entrypoint] No explicit multi-accelerator list detected, starting single backend instance (${accelerator})"
start_backend_process "$accelerator" "" "$single_port" "$single_log_file" "$bind_host"
backend_pids+=("$!")
else
if [[ ${#valid_devices[@]} -eq 1 ]]; then
echo "[entrypoint] Single accelerator device detected (${valid_devices[0]}), starting one backend instance (${accelerator})"
local single_gpu_port="$base_port"
local single_gpu_log_file="${default_log_file%.log}-gpu0.log"
backend_ports+=("$single_gpu_port")
start_backend_process "$accelerator" "${valid_devices[0]}" "$single_gpu_port" "$single_gpu_log_file" "$bind_host"
backend_pids+=("$!")
else
echo "[entrypoint] Multi-accelerator detected, starting one instance per device (${accelerator}): ${valid_devices[*]}"
for dev in "${valid_devices[@]}"; do
local port=$((base_port + idx))
local instance_log_file="${default_log_file%.log}-gpu${idx}.log"
backend_ports+=("$port")
echo "[entrypoint] Starting ASR instance #${idx} on ${accelerator} device ${dev}, bind ${bind_host}:${port}"
WORKERS="1" start_backend_process "$accelerator" "$dev" "$port" "$instance_log_file" "$bind_host"
backend_pids+=("$!")
idx=$((idx + 1))
done
fi
fi
local port
for port in "${backend_ports[@]}"; do
if ! wait_for_port "$bind_host" "$port" "$ready_timeout"; then
echo "[entrypoint] Backend instance on ${bind_host}:${port} failed to become ready in ${ready_timeout}s"
for pid in "${backend_pids[@]}"; do
kill "$pid" >/dev/null 2>&1 || true
done
wait >/dev/null 2>&1 || true
exit 1
fi
done
local nginx_conf="/tmp/qwen3-asr-internal-nginx.conf"
{
echo "worker_processes auto;"
echo "events { worker_connections 1024; }"
echo "http {"
echo " limit_req_status 429;"
if is_positive_integer "$rate_limit_rps"; then
# Global token bucket for the whole service, not per-client-IP.
echo " limit_req_zone \$server_name zone=api_rps:10m rate=${rate_limit_rps}r/s;"
fi
echo
echo " upstream qwen3_asr_upstream {"
echo " least_conn;"
for port in "${backend_ports[@]}"; do
echo " server ${bind_host}:${port} max_fails=3 fail_timeout=10s;"
done
echo " keepalive 128;"
echo " }"
echo
echo " map \$http_upgrade \$connection_upgrade {"
echo " default upgrade;"
echo " '' close;"
echo " }"
echo
echo " server {"
echo " listen ${public_port};"
echo " server_name _;"
echo " client_max_body_size 2048m;"
echo
echo " location / {"
if is_positive_integer "$rate_limit_rps"; then
echo " limit_req zone=api_rps burst=${rate_limit_burst} nodelay;"
fi
echo " proxy_pass http://qwen3_asr_upstream;"
echo " proxy_http_version 1.1;"
echo " proxy_set_header Host \$host;"
echo " proxy_set_header X-Real-IP \$remote_addr;"
echo " proxy_set_header X-Forwarded-For \$proxy_add_x_forwarded_for;"
echo " proxy_set_header X-Forwarded-Proto \$scheme;"
echo " proxy_set_header Upgrade \$http_upgrade;"
echo " proxy_set_header Connection \$connection_upgrade;"
echo " proxy_connect_timeout 10s;"
echo " proxy_send_timeout 3600s;"
echo " proxy_read_timeout 3600s;"
echo " send_timeout 3600s;"
echo " proxy_buffering off;"
echo " }"
echo " }"
echo "}"
} > "$nginx_conf"
local nginx_pid=""
cleanup() {
set +e
if [[ -n "$nginx_pid" ]]; then
kill "$nginx_pid" >/dev/null 2>&1 || true
fi
for pid in "${backend_pids[@]}"; do
kill "$pid" >/dev/null 2>&1 || true
done
wait >/dev/null 2>&1 || true
}
trap cleanup EXIT INT TERM
echo "[entrypoint] Starting internal nginx load balancer on :${public_port}"
nginx -c "$nginx_conf" -g "daemon off;" &
nginx_pid="$!"
while true; do
if ! kill -0 "$nginx_pid" >/dev/null 2>&1; then
echo "[entrypoint] nginx exited unexpectedly"
exit 1
fi
for pid in "${backend_pids[@]}"; do
if ! kill -0 "$pid" >/dev/null 2>&1; then
echo "[entrypoint] backend process ${pid} exited unexpectedly"
exit 1
fi
done
sleep 2
done
}
has_default_cmd=false
if [[ $# -eq 2 && "$1" == "$DEFAULT_CMD_1" && "$2" == "$DEFAULT_CMD_2" ]]; then
has_default_cmd=true
fi
# If user passes a custom command, respect it and bypass auto multi-GPU logic.
if [[ $# -gt 0 && "$has_default_cmd" != "true" ]]; then
exec "$@"
fi
ACTIVE_ACCELERATOR="$(detect_accelerator)"
devices_csv="$(normalize_visible_devices "$ACTIVE_ACCELERATOR")"
topology="$(normalize_topology)"
case "$topology" in
isolated)
export ASR_ACTIVE_TOPOLOGY="isolated"
start_internal_nginx_mode "$devices_csv"
;;
sharded)
if ! start_sharded_backend_mode "$ACTIVE_ACCELERATOR" "$devices_csv"; then
export ASR_ACTIVE_TOPOLOGY="isolated"
start_internal_nginx_mode "$devices_csv"
fi
;;
auto)
if start_sharded_backend_mode "$ACTIVE_ACCELERATOR" "$devices_csv"; then
exit 0
fi
export ASR_ACTIVE_TOPOLOGY="isolated"
start_internal_nginx_mode "$devices_csv"
;;
esac