#!/usr/bin/env bash set -euo pipefail DEFAULT_CMD_1="${APP_PYTHON_BIN:-/opt/venv/bin/python}" DEFAULT_CMD_2="start.py" ACTIVE_ACCELERATOR="" normalize_topology() { local raw="${ASR_DEPLOY_TOPOLOGY:-isolated}" raw="${raw,,}" case "$raw" in isolated|sharded|auto) echo "$raw" return 0 ;; *) echo "[entrypoint] Invalid ASR_DEPLOY_TOPOLOGY=${raw}, fallback to isolated" >&2 echo "isolated" return 0 ;; esac } detect_accelerator() { local configured="${ACCELERATOR:-auto}" configured="${configured,,}" case "$configured" in nvidia|metax|iluvatar|mthreads|cpu) echo "$configured" return 0 ;; cuda) echo "nvidia" return 0 ;; maca|muxi|mx) echo "metax" return 0 ;; ix|tianshu|天数) echo "iluvatar" return 0 ;; mthreads|musa|moorethreads|摩尔线程) echo "mthreads" return 0 ;; esac if command -v mx-smi >/dev/null 2>&1; then echo "metax" return 0 fi if command -v ixsmi >/dev/null 2>&1; then echo "iluvatar" return 0 fi if command -v mthreads-gmi >/dev/null 2>&1; then echo "mthreads" return 0 fi if command -v nvidia-smi >/dev/null 2>&1; then echo "nvidia" return 0 fi echo "cpu" } list_nvidia_devices() { if command -v nvidia-smi >/dev/null 2>&1; then nvidia-smi --query-gpu=index --format=csv,noheader,nounits | sed '/^$/d' fi } list_metax_devices() { if ! command -v mx-smi >/dev/null 2>&1; then return 0 fi local output output="$(mx-smi -L 2>/dev/null || true)" if [[ -n "$output" ]]; then echo "$output" | awk 'NF { print NR - 1 }' return 0 fi output="$(mx-smi 2>/dev/null || true)" if [[ -n "$output" ]]; then echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }' fi } list_iluvatar_devices() { if ! command -v ixsmi >/dev/null 2>&1; then return 0 fi local output output="$(ixsmi -L 2>/dev/null || true)" if [[ -n "$output" ]]; then echo "$output" | awk 'NF { print NR - 1 }' return 0 fi output="$(ixsmi 2>/dev/null || true)" if [[ -n "$output" ]]; then echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }' fi } list_mthreads_devices() { if ! command -v mthreads-gmi >/dev/null 2>&1; then return 0 fi local output output="$(mthreads-gmi -L 2>/dev/null || true)" if [[ -n "$output" ]]; then echo "$output" | awk 'NF { print NR - 1 }' return 0 fi output="$(mthreads-gmi list 2>/dev/null || true)" if [[ -n "$output" ]]; then echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }' return 0 fi output="$(mthreads-gmi 2>/dev/null || true)" if [[ -n "$output" ]]; then echo "$output" | awk '/^[[:space:]]*\|[[:space:]]*[0-9]+[[:space:]]/ { print $2 }' fi } normalize_visible_devices() { local accelerator="$1" local shared="${ASR_VISIBLE_DEVICES:-}" shared="${shared// /}" if [[ -n "$shared" && "$shared" != "none" && "$shared" != "void" ]]; then if [[ "$shared" == "all" ]]; then case "$accelerator" in nvidia) mapfile -t gpu_indexes < <(list_nvidia_devices) ;; metax) mapfile -t gpu_indexes < <(list_metax_devices) ;; iluvatar) mapfile -t gpu_indexes < <(list_iluvatar_devices) ;; mthreads) mapfile -t gpu_indexes < <(list_mthreads_devices) ;; *) gpu_indexes=() ;; esac if [[ ${#gpu_indexes[@]} -eq 0 ]]; then echo "" return 0 fi local shared_joined shared_joined=$(IFS=,; echo "${gpu_indexes[*]}") echo "$shared_joined" return 0 fi echo "$shared" return 0 fi local raw="" case "$accelerator" in nvidia) raw="${CUDA_VISIBLE_DEVICES:-}" ;; metax) raw="${METAX_VISIBLE_DEVICES:-${MACA_VISIBLE_DEVICES:-${MX_VISIBLE_DEVICES:-}}}" ;; iluvatar) raw="${ILUVATAR_VISIBLE_DEVICES:-${IX_VISIBLE_DEVICES:-${CUDA_VISIBLE_DEVICES:-}}}" ;; mthreads) raw="${MTHREADS_VISIBLE_DEVICES:-${MUSA_VISIBLE_DEVICES:-${CUDA_VISIBLE_DEVICES:-}}}" ;; *) echo "" return 0 ;; esac raw="${raw// /}" if [[ -z "$raw" || "$raw" == "none" || "$raw" == "void" ]]; then echo "" return 0 fi if [[ "$raw" == "all" ]]; then case "$accelerator" in nvidia) mapfile -t gpu_indexes < <(list_nvidia_devices) ;; metax) mapfile -t gpu_indexes < <(list_metax_devices) ;; iluvatar) mapfile -t gpu_indexes < <(list_iluvatar_devices) ;; mthreads) mapfile -t gpu_indexes < <(list_mthreads_devices) ;; esac if [[ ${#gpu_indexes[@]} -eq 0 ]]; then echo "" return 0 fi local joined joined=$(IFS=,; echo "${gpu_indexes[*]}") echo "$joined" return 0 fi echo "$raw" } export_visible_devices_aliases() { local accelerator="$1" local devices_csv="$2" if [[ -z "$devices_csv" ]]; then return 0 fi export ASR_VISIBLE_DEVICES="$devices_csv" case "$accelerator" in nvidia) export CUDA_VISIBLE_DEVICES="$devices_csv" ;; metax) export METAX_VISIBLE_DEVICES="$devices_csv" export MACA_VISIBLE_DEVICES="$devices_csv" export MX_VISIBLE_DEVICES="$devices_csv" ;; iluvatar) export ILUVATAR_VISIBLE_DEVICES="$devices_csv" export IX_VISIBLE_DEVICES="$devices_csv" export CUDA_VISIBLE_DEVICES="$devices_csv" ;; mthreads) export MTHREADS_VISIBLE_DEVICES="$devices_csv" export MUSA_VISIBLE_DEVICES="$devices_csv" export CUDA_VISIBLE_DEVICES="$devices_csv" ;; esac } supports_sharded_topology() { local accelerator="$1" case "$accelerator" in nvidia|metax|iluvatar|mthreads) return 0 ;; *) return 1 ;; esac } start_backend_process() { local accelerator="$1" local device="$2" local port="$3" local log_file="$4" local bind_host="$5" local workers="${WORKERS:-1}" if [[ "$accelerator" == "nvidia" && -n "$device" ]]; then CUDA_VISIBLE_DEVICES="$device" \ ACCELERATOR="nvidia" \ DEVICE="cuda:0" \ WORKERS="$workers" \ HOST="$bind_host" \ PORT="$port" \ LOG_FILE="$log_file" \ "$DEFAULT_CMD_1" "$DEFAULT_CMD_2" & return 0 fi if [[ "$accelerator" == "metax" && -n "$device" ]]; then METAX_VISIBLE_DEVICES="$device" \ MACA_VISIBLE_DEVICES="$device" \ MX_VISIBLE_DEVICES="$device" \ ACCELERATOR="metax" \ DEVICE="cuda:0" \ WORKERS="$workers" \ HOST="$bind_host" \ PORT="$port" \ LOG_FILE="$log_file" \ "$DEFAULT_CMD_1" "$DEFAULT_CMD_2" & return 0 fi if [[ "$accelerator" == "iluvatar" && -n "$device" ]]; then ILUVATAR_VISIBLE_DEVICES="$device" \ IX_VISIBLE_DEVICES="$device" \ CUDA_VISIBLE_DEVICES="$device" \ ACCELERATOR="iluvatar" \ DEVICE="cuda:0" \ WORKERS="$workers" \ HOST="$bind_host" \ PORT="$port" \ LOG_FILE="$log_file" \ "$DEFAULT_CMD_1" "$DEFAULT_CMD_2" & return 0 fi if [[ "$accelerator" == "mthreads" && -n "$device" ]]; then MTHREADS_VISIBLE_DEVICES="$device" \ MUSA_VISIBLE_DEVICES="$device" \ CUDA_VISIBLE_DEVICES="$device" \ ACCELERATOR="mthreads" \ DEVICE="cuda:0" \ WORKERS="$workers" \ HOST="$bind_host" \ PORT="$port" \ LOG_FILE="$log_file" \ "$DEFAULT_CMD_1" "$DEFAULT_CMD_2" & return 0 fi ACCELERATOR="$accelerator" \ WORKERS="$workers" \ HOST="$bind_host" \ PORT="$port" \ LOG_FILE="$log_file" \ "$DEFAULT_CMD_1" "$DEFAULT_CMD_2" & } exec_backend_process() { local accelerator="$1" local device="$2" local port="$3" local log_file="$4" local bind_host="$5" local workers="${WORKERS:-1}" if [[ "$accelerator" == "nvidia" && -n "$device" ]]; then export CUDA_VISIBLE_DEVICES="$device" export DEVICE="cuda:0" elif [[ "$accelerator" == "metax" && -n "$device" ]]; then export METAX_VISIBLE_DEVICES="$device" export MACA_VISIBLE_DEVICES="$device" export MX_VISIBLE_DEVICES="$device" export DEVICE="cuda:0" elif [[ "$accelerator" == "iluvatar" && -n "$device" ]]; then export ILUVATAR_VISIBLE_DEVICES="$device" export IX_VISIBLE_DEVICES="$device" export CUDA_VISIBLE_DEVICES="$device" export DEVICE="cuda:0" elif [[ "$accelerator" == "mthreads" && -n "$device" ]]; then export MTHREADS_VISIBLE_DEVICES="$device" export MUSA_VISIBLE_DEVICES="$device" export CUDA_VISIBLE_DEVICES="$device" export DEVICE="cuda:0" fi export ACCELERATOR="$accelerator" export WORKERS="$workers" export HOST="$bind_host" export PORT="$port" export LOG_FILE="$log_file" exec "$DEFAULT_CMD_1" "$DEFAULT_CMD_2" } start_sharded_backend_mode() { local accelerator="$1" local devices_csv="$2" local bind_host="${HOST:-0.0.0.0}" local public_port="${PORT:-8000}" local log_file="${LOG_FILE:-/app/data/logs/qwen3-asr.log}" local workers="${WORKERS:-1}" local shard_size=0 if [[ -n "$devices_csv" ]]; then IFS=',' read -r -a devs <<< "$devices_csv" shard_size="${#devs[@]}" fi if ! supports_sharded_topology "$accelerator"; then echo "[entrypoint] accelerator=${accelerator} does not support sharded topology, fallback to isolated" return 1 fi if [[ -z "$devices_csv" ]]; then echo "[entrypoint] sharded topology requires visible GPU devices, fallback to isolated" return 1 fi if ! [[ "$shard_size" =~ ^[0-9]+$ ]] || (( shard_size < 2 )); then echo "[entrypoint] sharded topology requires at least 2 devices, got shard_size=${shard_size}, fallback to isolated" return 1 fi echo "[entrypoint] Starting sharded backend: accelerator=${accelerator} devices=${devices_csv} shard_size=${shard_size} bind=${bind_host}:${public_port}" export ASR_ACTIVE_TOPOLOGY="sharded" export_visible_devices_aliases "$accelerator" "$devices_csv" exec_backend_process "$accelerator" "" "$public_port" "$log_file" "$bind_host" } wait_for_port() { local host="$1" local port="$2" local timeout_sec="$3" local deadline=$((SECONDS + timeout_sec)) while (( SECONDS < deadline )); do if (echo >/dev/tcp/"$host"/"$port") >/dev/null 2>&1; then return 0 fi sleep 1 done return 1 } is_positive_integer() { [[ "$1" =~ ^[0-9]+$ ]] && (( "$1" > 0 )) } start_internal_nginx_mode() { local devices_csv="$1" local accelerator="${ACTIVE_ACCELERATOR:-auto}" local bind_host="${MULTI_GPU_BIND_HOST:-127.0.0.1}" local base_port="18000" local public_port="${PORT:-8000}" local ready_timeout="${MULTI_GPU_READY_TIMEOUT:-180}" local rate_limit_rps="${NGINX_RATE_LIMIT_RPS:-0}" local rate_limit_burst="${NGINX_RATE_LIMIT_BURST:-0}" local valid_devices=() local devices=() if [[ -n "$devices_csv" ]]; then IFS=',' read -r -a devices <<< "$devices_csv" for dev in "${devices[@]}"; do if [[ -n "$dev" ]]; then valid_devices+=("$dev") fi done fi if [[ "$rate_limit_rps" != "0" ]] && ! is_positive_integer "$rate_limit_rps"; then echo "[entrypoint] Invalid NGINX_RATE_LIMIT_RPS=${rate_limit_rps}, fallback to 0 (disabled)" rate_limit_rps="0" fi if [[ "$rate_limit_burst" != "0" ]] && ! is_positive_integer "$rate_limit_burst"; then echo "[entrypoint] Invalid NGINX_RATE_LIMIT_BURST=${rate_limit_burst}, fallback to 0" rate_limit_burst="0" fi if [[ "$rate_limit_rps" != "0" && "$rate_limit_burst" == "0" ]]; then # By default, give short burst headroom equal to rate limit. rate_limit_burst="$rate_limit_rps" fi if ! command -v nginx >/dev/null 2>&1; then if [[ ${#valid_devices[@]} -le 1 ]]; then local direct_device="" local direct_log_file="${LOG_FILE:-/app/data/logs/qwen3-asr.log}" local direct_bind_host="${HOST:-0.0.0.0}" if [[ ${#valid_devices[@]} -eq 1 ]]; then direct_device="${valid_devices[0]}" fi echo "[entrypoint] nginx not found; starting single backend directly on ${direct_bind_host}:${public_port}" exec_backend_process "$accelerator" "$direct_device" "$public_port" "$direct_log_file" "$direct_bind_host" fi echo "[entrypoint] nginx not found in image; cannot start multi-backend proxy mode" exit 1 fi local backend_ports=() local backend_pids=() local idx=0 local default_log_file="${LOG_FILE:-/app/data/logs/qwen3-asr.log}" if [[ ${#valid_devices[@]} -eq 0 ]]; then local single_port="$base_port" local single_log_file="${default_log_file%.log}-gpu0.log" backend_ports+=("$single_port") echo "[entrypoint] No explicit multi-accelerator list detected, starting single backend instance (${accelerator})" start_backend_process "$accelerator" "" "$single_port" "$single_log_file" "$bind_host" backend_pids+=("$!") else if [[ ${#valid_devices[@]} -eq 1 ]]; then echo "[entrypoint] Single accelerator device detected (${valid_devices[0]}), starting one backend instance (${accelerator})" local single_gpu_port="$base_port" local single_gpu_log_file="${default_log_file%.log}-gpu0.log" backend_ports+=("$single_gpu_port") start_backend_process "$accelerator" "${valid_devices[0]}" "$single_gpu_port" "$single_gpu_log_file" "$bind_host" backend_pids+=("$!") else echo "[entrypoint] Multi-accelerator detected, starting one instance per device (${accelerator}): ${valid_devices[*]}" for dev in "${valid_devices[@]}"; do local port=$((base_port + idx)) local instance_log_file="${default_log_file%.log}-gpu${idx}.log" backend_ports+=("$port") echo "[entrypoint] Starting ASR instance #${idx} on ${accelerator} device ${dev}, bind ${bind_host}:${port}" WORKERS="1" start_backend_process "$accelerator" "$dev" "$port" "$instance_log_file" "$bind_host" backend_pids+=("$!") idx=$((idx + 1)) done fi fi local port for port in "${backend_ports[@]}"; do if ! wait_for_port "$bind_host" "$port" "$ready_timeout"; then echo "[entrypoint] Backend instance on ${bind_host}:${port} failed to become ready in ${ready_timeout}s" for pid in "${backend_pids[@]}"; do kill "$pid" >/dev/null 2>&1 || true done wait >/dev/null 2>&1 || true exit 1 fi done local nginx_conf="/tmp/qwen3-asr-internal-nginx.conf" { echo "worker_processes auto;" echo "events { worker_connections 1024; }" echo "http {" echo " limit_req_status 429;" if is_positive_integer "$rate_limit_rps"; then # Global token bucket for the whole service, not per-client-IP. echo " limit_req_zone \$server_name zone=api_rps:10m rate=${rate_limit_rps}r/s;" fi echo echo " upstream qwen3_asr_upstream {" echo " least_conn;" for port in "${backend_ports[@]}"; do echo " server ${bind_host}:${port} max_fails=3 fail_timeout=10s;" done echo " keepalive 128;" echo " }" echo echo " map \$http_upgrade \$connection_upgrade {" echo " default upgrade;" echo " '' close;" echo " }" echo echo " server {" echo " listen ${public_port};" echo " server_name _;" echo " client_max_body_size 2048m;" echo echo " location / {" if is_positive_integer "$rate_limit_rps"; then echo " limit_req zone=api_rps burst=${rate_limit_burst} nodelay;" fi echo " proxy_pass http://qwen3_asr_upstream;" echo " proxy_http_version 1.1;" echo " proxy_set_header Host \$host;" echo " proxy_set_header X-Real-IP \$remote_addr;" echo " proxy_set_header X-Forwarded-For \$proxy_add_x_forwarded_for;" echo " proxy_set_header X-Forwarded-Proto \$scheme;" echo " proxy_set_header Upgrade \$http_upgrade;" echo " proxy_set_header Connection \$connection_upgrade;" echo " proxy_connect_timeout 10s;" echo " proxy_send_timeout 3600s;" echo " proxy_read_timeout 3600s;" echo " send_timeout 3600s;" echo " proxy_buffering off;" echo " }" echo " }" echo "}" } > "$nginx_conf" local nginx_pid="" cleanup() { set +e if [[ -n "$nginx_pid" ]]; then kill "$nginx_pid" >/dev/null 2>&1 || true fi for pid in "${backend_pids[@]}"; do kill "$pid" >/dev/null 2>&1 || true done wait >/dev/null 2>&1 || true } trap cleanup EXIT INT TERM echo "[entrypoint] Starting internal nginx load balancer on :${public_port}" nginx -c "$nginx_conf" -g "daemon off;" & nginx_pid="$!" while true; do if ! kill -0 "$nginx_pid" >/dev/null 2>&1; then echo "[entrypoint] nginx exited unexpectedly" exit 1 fi for pid in "${backend_pids[@]}"; do if ! kill -0 "$pid" >/dev/null 2>&1; then echo "[entrypoint] backend process ${pid} exited unexpectedly" exit 1 fi done sleep 2 done } has_default_cmd=false if [[ $# -eq 2 && "$1" == "$DEFAULT_CMD_1" && "$2" == "$DEFAULT_CMD_2" ]]; then has_default_cmd=true fi # If user passes a custom command, respect it and bypass auto multi-GPU logic. if [[ $# -gt 0 && "$has_default_cmd" != "true" ]]; then exec "$@" fi ACTIVE_ACCELERATOR="$(detect_accelerator)" devices_csv="$(normalize_visible_devices "$ACTIVE_ACCELERATOR")" topology="$(normalize_topology)" case "$topology" in isolated) export ASR_ACTIVE_TOPOLOGY="isolated" start_internal_nginx_mode "$devices_csv" ;; sharded) if ! start_sharded_backend_mode "$ACTIVE_ACCELERATOR" "$devices_csv"; then export ASR_ACTIVE_TOPOLOGY="isolated" start_internal_nginx_mode "$devices_csv" fi ;; auto) if start_sharded_backend_mode "$ACTIVE_ACCELERATOR" "$devices_csv"; then exit 0 fi export ASR_ACTIVE_TOPOLOGY="isolated" start_internal_nginx_mode "$devices_csv" ;; esac