name: followup-audio-asr services: asr: image: ${ASR_IMAGE:?Run build-image.sh and init-env.py first} pull_policy: never container_name: followup-audio-asr restart: unless-stopped stop_grace_period: 660s init: true ports: - "127.0.0.1:8001:8000" environment: CUDA_VISIBLE_DEVICES: "0" VLLM_API_KEY: ${ASR_API_KEY:?Initialize the private .env first} HF_HUB_OFFLINE: "1" TRANSFORMERS_OFFLINE: "1" HF_HUB_DISABLE_TELEMETRY: "1" DO_NOT_TRACK: "1" VLLM_NO_USAGE_STATS: "1" OMP_NUM_THREADS: "4" volumes: - /home/www/qwen-vllm/asr/models/Qwen3-ASR-1.7B:/model:ro shm_size: "2gb" cap_drop: [ALL] security_opt: [no-new-privileges:true] deploy: resources: reservations: devices: - driver: nvidia device_ids: ["1"] capabilities: [gpu] command: - /model - --served-model-name - Qwen/Qwen3-ASR-1.7B - --host - 0.0.0.0 - --port - "8000" - --dtype - bfloat16 - --tensor-parallel-size - "1" - --gpu-memory-utilization - ${ASR_GPU_MEMORY_UTILIZATION:-0.15} - --max-num-seqs - "1" - --max-model-len - "32768" - --max-num-batched-tokens - "4096" - --shutdown-timeout - "600" - --enforce-eager - --no-enable-log-requests healthcheck: test: [CMD, python3, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5).read()"] interval: 20s timeout: 8s start_period: 180s retries: 6 logging: driver: json-file options: max-size: "10m" max-file: "3"