ops: add isolated ASR deployment and safe GPU test handoff
This commit is contained in:
@@ -0,0 +1,67 @@
|
||||
name: followup-audio-asr
|
||||
services:
|
||||
asr:
|
||||
image: ${ASR_IMAGE:?Run build-image.sh and init-env.py first}
|
||||
pull_policy: never
|
||||
container_name: followup-audio-asr
|
||||
restart: unless-stopped
|
||||
stop_grace_period: 660s
|
||||
init: true
|
||||
ports:
|
||||
- "127.0.0.1:8001:8000"
|
||||
environment:
|
||||
CUDA_VISIBLE_DEVICES: "0"
|
||||
VLLM_API_KEY: ${ASR_API_KEY:?Initialize the private .env first}
|
||||
HF_HUB_OFFLINE: "1"
|
||||
TRANSFORMERS_OFFLINE: "1"
|
||||
HF_HUB_DISABLE_TELEMETRY: "1"
|
||||
DO_NOT_TRACK: "1"
|
||||
VLLM_NO_USAGE_STATS: "1"
|
||||
OMP_NUM_THREADS: "4"
|
||||
volumes:
|
||||
- /home/www/qwen-vllm/asr/models/Qwen3-ASR-1.7B:/model:ro
|
||||
shm_size: "2gb"
|
||||
cap_drop: [ALL]
|
||||
security_opt: [no-new-privileges:true]
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids: ["1"]
|
||||
capabilities: [gpu]
|
||||
command:
|
||||
- /model
|
||||
- --served-model-name
|
||||
- Qwen/Qwen3-ASR-1.7B
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --dtype
|
||||
- bfloat16
|
||||
- --tensor-parallel-size
|
||||
- "1"
|
||||
- --gpu-memory-utilization
|
||||
- ${ASR_GPU_MEMORY_UTILIZATION:-0.15}
|
||||
- --max-num-seqs
|
||||
- "1"
|
||||
- --max-model-len
|
||||
- "32768"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --shutdown-timeout
|
||||
- "600"
|
||||
- --enforce-eager
|
||||
- --no-enable-log-requests
|
||||
healthcheck:
|
||||
test: [CMD, python3, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5).read()"]
|
||||
interval: 20s
|
||||
timeout: 8s
|
||||
start_period: 180s
|
||||
retries: 6
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: "10m"
|
||||
max-file: "3"
|
||||
Reference in New Issue
Block a user