Files
zyt/deployment/followup-audio-asr/compose.yaml
T

68 lines
1.7 KiB
YAML

name: followup-audio-asr
services:
asr:
image: ${ASR_IMAGE:?Run build-image.sh and init-env.py first}
pull_policy: never
container_name: followup-audio-asr
restart: unless-stopped
stop_grace_period: 660s
init: true
ports:
- "127.0.0.1:8001:8000"
environment:
CUDA_VISIBLE_DEVICES: "0"
VLLM_API_KEY: ${ASR_API_KEY:?Initialize the private .env first}
HF_HUB_OFFLINE: "1"
TRANSFORMERS_OFFLINE: "1"
HF_HUB_DISABLE_TELEMETRY: "1"
DO_NOT_TRACK: "1"
VLLM_NO_USAGE_STATS: "1"
OMP_NUM_THREADS: "4"
volumes:
- /home/www/qwen-vllm/asr/models/Qwen3-ASR-1.7B:/model:ro
shm_size: "2gb"
cap_drop: [ALL]
security_opt: [no-new-privileges:true]
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["1"]
capabilities: [gpu]
command:
- /model
- --served-model-name
- Qwen/Qwen3-ASR-1.7B
- --host
- 0.0.0.0
- --port
- "8000"
- --dtype
- bfloat16
- --tensor-parallel-size
- "1"
- --gpu-memory-utilization
- ${ASR_GPU_MEMORY_UTILIZATION:-0.15}
- --max-num-seqs
- "1"
- --max-model-len
- "32768"
- --max-num-batched-tokens
- "4096"
- --shutdown-timeout
- "600"
- --enforce-eager
- --no-enable-log-requests
healthcheck:
test: [CMD, python3, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5).read()"]
interval: 20s
timeout: 8s
start_period: 180s
retries: 6
logging:
driver: json-file
options:
max-size: "10m"
max-file: "3"