diff --git a/deployment/followup-audio-asr/.dockerignore b/deployment/followup-audio-asr/.dockerignore new file mode 100644 index 000000000..a83dbf3f7 --- /dev/null +++ b/deployment/followup-audio-asr/.dockerignore @@ -0,0 +1,12 @@ +* +!Dockerfile +!audio-requirements.txt +!cpu_preflight.py + +__pycache__/ + +*.pyc + +acceptance/ + +.gpu1-mode* diff --git a/deployment/followup-audio-asr/.env.example b/deployment/followup-audio-asr/.env.example new file mode 100644 index 000000000..b805b36fb --- /dev/null +++ b/deployment/followup-audio-asr/.env.example @@ -0,0 +1,4 @@ +# Copying this placeholder is insufficient: init-env.py creates random secret .env. +ASR_IMAGE=sha256:REPLACE_WITH_BUILT_IMAGE_ID +ASR_API_KEY=REPLACE_WITH_RANDOM_SECRET +ASR_GPU_MEMORY_UTILIZATION=0.15 diff --git a/deployment/followup-audio-asr/.gitignore b/deployment/followup-audio-asr/.gitignore new file mode 100644 index 000000000..5af2cb9cc --- /dev/null +++ b/deployment/followup-audio-asr/.gitignore @@ -0,0 +1,12 @@ +.env +*.log +image-id.txt +fixtures/ + +__pycache__/ + +*.pyc + +acceptance/ + +.gpu1-mode* diff --git a/deployment/followup-audio-asr/Dockerfile b/deployment/followup-audio-asr/Dockerfile new file mode 100644 index 000000000..3aa13b26f --- /dev/null +++ b/deployment/followup-audio-asr/Dockerfile @@ -0,0 +1,7 @@ +ARG BASE_IMAGE=local/qwen-asr-base:20261008-251eba5cc7c1 +FROM ${BASE_IMAGE} +LABEL org.opencontainers.image.description="Isolated Qwen3-ASR audio dependencies" \ + local.base_image_id="sha256:251eba5cc7c12fed0b75da22a9240e582b1c9e39f6fbc064f86781b963bd814f" +COPY audio-requirements.txt /opt/qwen-asr/audio-requirements.txt +RUN python3 -m pip install --no-cache-dir --no-deps --only-binary=:all: -r /opt/qwen-asr/audio-requirements.txt +COPY cpu_preflight.py /opt/qwen-asr/cpu_preflight.py diff --git a/deployment/followup-audio-asr/README.md b/deployment/followup-audio-asr/README.md new file mode 100644 index 000000000..898eb2152 --- /dev/null +++ b/deployment/followup-audio-asr/README.md @@ -0,0 +1,116 @@ +# Qwen3-ASR isolated preparation package + +This package prepares a **separate** ASR image and immutable model snapshot. It does +not change, replace, stop, or restart the existing `qwen36-35b` deployment. +No GPU process is started by `download_model.py`, `build-image.sh`, or `init-env.py`. + +## Fixed identities + +- Base image ID: `sha256:251eba5cc7c12fed0b75da22a9240e582b1c9e39f6fbc064f86781b963bd814f` +- Base: vLLM 0.24.0, torch 2.11.0+cu130, transformers 5.12.1, numpy 2.2.6. +- Official model: `Qwen/Qwen3-ASR-1.7B`. +- Official ModelScope snapshot: `a04930dbe5419bfee073f7cade734f572689a3a8`. +- Weights total 4,698,521,512 bytes; do not use the earlier ~3.5GB estimate. +- New service root: `/home/www/qwen-vllm/asr`. +- Model root: `/home/www/qwen-vllm/asr/models/Qwen3-ASR-1.7B`. +- `DOWNLOAD_MANIFEST.json` preserves official SHA256 and size for every payload. +- Runtime image is pinned to the built immutable image ID in private `.env`. + +The existing base contains the Qwen3ASR model registry and transcription implementation +but lacks audio extras. The independent thin image adds only `av`, `scipy`, +`soundfile`, and `soxr` with explicit versions and `--no-deps`; it preserves torch, +transformers, numpy and the original image bytes. CPU preflight performs actual WAV +encode/decode, resampling, and Whisper feature extraction plus static route checks. +This is **not** GPU inference or endpoint acceptance. + +## Prepare without GPU execution + +```sh +cd /home/www/qwen-vllm/asr +python3 download_model.py /home/www/qwen-vllm/asr/models/Qwen3-ASR-1.7B +./build-image.sh +python3 init-env.py +docker compose --env-file .env -f compose.yaml config --quiet +``` + +`init-env.py` creates a random API key in a new mode-0600 `.env`, refuses to overwrite +an existing file, and never prints the key. Do not publish `.env`, full compose +config output, container environment, or API Authorization headers. + +## Start only after the project controller has released GPU1 + +**The controller must first verify that Qwen is healthy on GPU0 alone, GPU1 has no +competing process, and the shared-card guard permits ASR. Preparation is not this gate.** + +```sh +cd /home/www/qwen-vllm/asr +docker compose --env-file .env -f compose.yaml up -d --no-build --pull never asr +``` + +Host GPU1 alone is exposed via Docker DeviceIDs `["1"]`; it appears as logical +CUDA device 0 inside the one-GPU container. Host port is only `127.0.0.1:8001`. +`restart: unless-stopped` supports reboot persistence without reactivating an +explicitly stopped service. Initial `gpu_memory_utilization=0.15` is adjustable +in `.env`; do not increase without measuring allocation and the shared-card budget. +`max-num-seqs=1`, `max-model-len=32768`, eager execution and max batched tokens 4096 +bound the initial test workload. The model native text context is 65536. The installed +vLLM audio encoder maps about 13 audio tokens/second: 745 seconds is about 9685 +audio tokens before output, so 4096 is not safe for an unsplit 7–12 minute input. +A 32768-token BF16 KV cache is approximately 3.5 GiB from the model configuration +(28 layers, 8 KV heads, head dimension 128); this is an estimate, not measured VRAM. +Keep the 0.15 allocation cap and verify actual startup/long-audio behavior. The +OpenAI transcription implementation has internal clip splitting based on the +feature extractor's 30-second clip limit, but the application should still segment +recordings into at most 5-minute bounded jobs with overlap and ordered merging. +An hour-long recording must be split; this package does not claim a one-hour +single-request acceptance. Long calls need a separate chunking/quality gate. +No forced aligner is loaded; word timestamps are not promised. + +## Verify after controller start + +```sh +cd /home/www/qwen-vllm/asr +python3 probe.py /absolute/path/to/non-sensitive-speech.wav +``` + +The probe requires `/health` HTTP 200, rejects unauthenticated `/v1/models` with +401, confirms the served model with authentication, and requires nonempty text +from an actual multipart `/v1/audio/transcriptions` request. The authenticated +probe reads the key locally and does not print it. A public or synthetic fixture is +required; never use patient recordings for this gate. Recognition accuracy needs +an expected-text comparison/listening in addition to a successful HTTP response. + +## Stop / rollback only the new ASR service + +```sh +cd /home/www/qwen-vllm/asr +docker compose --env-file .env -f compose.yaml stop asr +``` + +This releases ASR's GPU allocation while retaining model/image/config artifacts. +Do not start ComfyUI on GPU1 until the parent-controlled shared-card guard has +confirmed ASR stopped and memory released. ComfyUI is not deployed by this package. + +Official references: +- https://github.com/QwenLM/Qwen3-ASR#deployment-with-vllm +- https://modelscope.cn/models/Qwen/Qwen3-ASR-1.7B +- https://huggingface.co/Qwen/Qwen3-ASR-1.7B + + +## GPU 1 与未来 ComfyUI 测试的互斥使用 + +在 ai 服务器运行: + +```sh +python3 /home/www/qwen-vllm/asr/gpu1-mode.py status +# 等待 ASR 在途/排队请求结束并停止它,确认 GPU 1 空闲;不会启动或停止 ComfyUI。 +python3 /home/www/qwen-vllm/asr/gpu1-mode.py test +# 测试实例完全退出、GPU 1 空闲后恢复 ASR,并等待健康检查。 +python3 /home/www/qwen-vllm/asr/gpu1-mode.py asr +``` + +控制器遇到未知 GPU 1 进程会拒绝操作,不强杀。ComfyUI 测试实例尚未创建;未来应显式绑定 GPU 1,不修改当前 GPU 2/3 上的服务。请通过此控制器切换,而非同时启动两个占卡服务。`unless-stopped` 保留手工停止状态,新服务同时设置 600 秒引擎优雅关闭和 660 秒容器停止宽限。 + +Qwen 单卡切换的原始配置、候选配置、合成文本/长文本/图片检查与回滚控制器保存在 ai 的 `/home/www/qwen-vllm/rollouts/20261008-single-gpu/`。需要回退双卡时,先释放 GPU 1,再运行该目录的 `deploy.py rollback`;它拒绝覆盖不认识的配置或驱逐 GPU 1 上的其他进程。此操作会重启 Qwen,需维护窗口。源码回滚和服务器运行状态回滚不是同一件事。 + +控制器无 GPU 单元检查:`python3 deployment/followup-audio-asr/test_gpu1_mode.py`。真实启动、转录和卡交接结果请以本次部署证据为准,不把 CPU-only 准备检查当 GPU 运行成功。 diff --git a/deployment/followup-audio-asr/audio-requirements.txt b/deployment/followup-audio-asr/audio-requirements.txt new file mode 100644 index 000000000..f2340fb99 --- /dev/null +++ b/deployment/followup-audio-asr/audio-requirements.txt @@ -0,0 +1,5 @@ +# Small isolated audio layer; base torch/transformers/numpy must not be replaced. +av==16.0.1 +scipy==1.16.3 +soundfile==0.13.1 +soxr==1.0.0 diff --git a/deployment/followup-audio-asr/build-image.sh b/deployment/followup-audio-asr/build-image.sh new file mode 100755 index 000000000..97cd71152 --- /dev/null +++ b/deployment/followup-audio-asr/build-image.sh @@ -0,0 +1,13 @@ +#!/bin/sh +set -eu +cd "$(CDPATH= cd -- "$(dirname -- "$0")" && pwd)" +base=sha256:251eba5cc7c12fed0b75da22a9240e582b1c9e39f6fbc064f86781b963bd814f +base_tag=local/qwen-asr-base:20261008-251eba5cc7c1 +actual=$(docker image inspect "$base" --format '{{.Id}}') +[ "$actual" = "$base" ] || { echo 'Base image mismatch' >&2; exit 1; } +docker tag "$base" "$base_tag" +docker build --pull=false --build-arg BASE_IMAGE="$base_tag" -t local/qwen-asr:20261008 . +docker image inspect local/qwen-asr:20261008 --format '{{.Id}}' > image-id.txt +printf 'ASR_IMAGE='; cat image-id.txt +# Explicitly no --gpus. This checks imports and signal processing only. +docker run --rm --network none -e NVIDIA_VISIBLE_DEVICES=void --entrypoint python3 "$(cat image-id.txt)" /opt/qwen-asr/cpu_preflight.py diff --git a/deployment/followup-audio-asr/compose.yaml b/deployment/followup-audio-asr/compose.yaml new file mode 100644 index 000000000..7ca3f333e --- /dev/null +++ b/deployment/followup-audio-asr/compose.yaml @@ -0,0 +1,67 @@ +name: followup-audio-asr +services: + asr: + image: ${ASR_IMAGE:?Run build-image.sh and init-env.py first} + pull_policy: never + container_name: followup-audio-asr + restart: unless-stopped + stop_grace_period: 660s + init: true + ports: + - "127.0.0.1:8001:8000" + environment: + CUDA_VISIBLE_DEVICES: "0" + VLLM_API_KEY: ${ASR_API_KEY:?Initialize the private .env first} + HF_HUB_OFFLINE: "1" + TRANSFORMERS_OFFLINE: "1" + HF_HUB_DISABLE_TELEMETRY: "1" + DO_NOT_TRACK: "1" + VLLM_NO_USAGE_STATS: "1" + OMP_NUM_THREADS: "4" + volumes: + - /home/www/qwen-vllm/asr/models/Qwen3-ASR-1.7B:/model:ro + shm_size: "2gb" + cap_drop: [ALL] + security_opt: [no-new-privileges:true] + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["1"] + capabilities: [gpu] + command: + - /model + - --served-model-name + - Qwen/Qwen3-ASR-1.7B + - --host + - 0.0.0.0 + - --port + - "8000" + - --dtype + - bfloat16 + - --tensor-parallel-size + - "1" + - --gpu-memory-utilization + - ${ASR_GPU_MEMORY_UTILIZATION:-0.15} + - --max-num-seqs + - "1" + - --max-model-len + - "32768" + - --max-num-batched-tokens + - "4096" + - --shutdown-timeout + - "600" + - --enforce-eager + - --no-enable-log-requests + healthcheck: + test: [CMD, python3, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5).read()"] + interval: 20s + timeout: 8s + start_period: 180s + retries: 6 + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" diff --git a/deployment/followup-audio-asr/cpu_preflight.py b/deployment/followup-audio-asr/cpu_preflight.py new file mode 100755 index 000000000..fe9f25079 --- /dev/null +++ b/deployment/followup-audio-asr/cpu_preflight.py @@ -0,0 +1,61 @@ +#!/usr/bin/env python3 +"""CPU-only dependency, format and route checks. Does not instantiate a model.""" +import importlib.metadata as metadata +import io +import json +from pathlib import Path +import numpy as np +import soundfile as sf +import soxr +import scipy +import av +import vllm +import torch +from transformers import WhisperFeatureExtractor + +signal=np.zeros(16000,dtype=np.float32) +encoded=io.BytesIO() +sf.write(encoded,signal,16000,format='WAV') +encoded.seek(0) +decoded,rate=sf.read(encoded,dtype='float32') +assert rate==16000 and decoded.shape==(16000,) +resampled=soxr.resample(decoded,16000,8000) +assert len(resampled)==8000 +features=WhisperFeatureExtractor()(decoded,sampling_rate=rate,return_tensors='np') +assert features['input_features'].shape[0]==1 +root=Path(vllm.__file__).parent +model_source=(root/'model_executor/models/qwen3_asr.py').read_text() +registry_source=(root/'model_executor/models/registry.py').read_text() +assert 'Qwen3ASRForConditionalGeneration' in registry_source +assert 'SupportsTranscription' in model_source +routes=[str(path.relative_to(root)) for path in (root/'entrypoints').rglob('*.py') if '"/v1/audio/transcriptions"' in path.read_text()] +assert routes, 'Transcription route missing' +assert not Path('/dev/nvidia0').exists(), 'This test must not have GPU device access' +assert vllm.__version__=='0.24.0' +assert metadata.version('numpy')=='2.2.6' +assert metadata.version('torch')=='2.11.0+cu130' +assert metadata.version('transformers')=='5.12.1' +print(json.dumps({'event':'cpu_preflight_pass','versions':{name:metadata.version(name) for name in ['vllm','torch','transformers','numpy','soundfile','scipy','soxr','av']},'wav_roundtrip_shape':list(decoded.shape),'resampled_shape':list(resampled.shape),'feature_shape':list(features['input_features'].shape),'transcription_route_files':routes,'gpu_device_access':False})) + +# Validate only the installed CLI option declarations. Constructing the full vLLM +# config on a no-GPU preflight host would try to infer a device. +import argparse +import ast +import inspect +from vllm.engine.arg_utils import AsyncEngineArgs +parser = argparse.ArgumentParser() +tree = ast.parse(Path(inspect.getfile(AsyncEngineArgs)).read_text()) +found = set() +for node in ast.walk(tree): + if (isinstance(node, ast.Call) and isinstance(node.func, ast.Attribute) + and isinstance(node.func.value, ast.Name) and node.func.value.id == "parser" + and node.func.attr == "add_argument" and node.args + and isinstance(node.args[0], ast.Constant) + and node.args[0].value in {"--enable-log-requests", "--shutdown-timeout"}): + eval(compile(ast.Expression(node), "", "eval"), + {"parser": parser, "argparse": argparse, "AsyncEngineArgs": AsyncEngineArgs, "int": int}) + found.add(node.args[0].value) +assert len(found) == 2 +args = parser.parse_args(["--no-enable-log-requests", "--shutdown-timeout", "600"]) +assert args.enable_log_requests is False and args.shutdown_timeout == 600 +print("CLI_PRIVACY_AND_SHUTDOWN_DECLARATIONS_PASS") diff --git a/deployment/followup-audio-asr/download_model.py b/deployment/followup-audio-asr/download_model.py new file mode 100755 index 000000000..6ed7a384e --- /dev/null +++ b/deployment/followup-audio-asr/download_model.py @@ -0,0 +1,72 @@ +#!/usr/bin/env python3 +"""Download a pinned official ModelScope snapshot; never replace mismatched files.""" +import hashlib +import json +import os +from pathlib import Path +import sys +import urllib.parse +import urllib.request + +MODEL = 'Qwen/Qwen3-ASR-1.7B' +REVISION = 'a04930dbe5419bfee073f7cade734f572689a3a8' +ROOT = Path(sys.argv[1] if len(sys.argv) > 1 else '/home/www/qwen-vllm/asr/models/Qwen3-ASR-1.7B') +API = f'https://modelscope.cn/api/v1/models/{MODEL}/repo' + +def digest(path): + h = hashlib.sha256() + with path.open('rb') as stream: + for chunk in iter(lambda: stream.read(8 * 1024 * 1024), b''): + h.update(chunk) + return h.hexdigest() + +ROOT.mkdir(parents=True, exist_ok=True) +url = API + '/files?' + urllib.parse.urlencode({'Revision': REVISION, 'Recursive': 'true'}) +with urllib.request.urlopen(url, timeout=60) as response: + listing = json.load(response) +assert listing['Code'] == 200 and listing['Success'], listing +files = [x for x in listing['Data']['Files'] if x['Type'] == 'blob' and x['Path'] != '.gitattributes'] +manifest = {'provider': 'ModelScope', 'model': MODEL, 'revision': REVISION, 'api': url, 'files': files} +manifest_path = ROOT / 'DOWNLOAD_MANIFEST.json' +if manifest_path.exists(): + old = json.loads(manifest_path.read_text()) + assert old == manifest, 'Existing manifest differs; preserve and abort' +else: + manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + '\n') +for row in files: + relative = Path(row['Path']) + assert not relative.is_absolute() and '..' not in relative.parts + path = ROOT / relative + if path.exists(): + assert path.stat().st_size == row['Size'] and digest(path) == row['Sha256'], f'Existing file mismatch: {path}' + print(json.dumps({'event': 'download_reused', 'path': str(path), 'sha256': row['Sha256'], 'bytes': row['Size']}), flush=True) + continue + path.parent.mkdir(parents=True, exist_ok=True) + partial = path.with_name(path.name + '.partial') + # A partial file is only a disposable incomplete transfer created by this script. + target = API + '?' + urllib.parse.urlencode({'Revision': REVISION, 'FilePath': row['Path']}) + print(json.dumps({'event': 'download_started', 'file': row['Path'], 'bytes': row['Size'], 'revision': REVISION}), flush=True) + h = hashlib.sha256() + size = 0 + with urllib.request.urlopen(target, timeout=120) as source, partial.open('wb') as output: + while True: + chunk = source.read(8 * 1024 * 1024) + if not chunk: + break + output.write(chunk) + h.update(chunk) + size += len(chunk) + assert size == row['Size'] and h.hexdigest() == row['Sha256'], f'Download integrity mismatch: {path}' + os.replace(partial, path) + print(json.dumps({'event': 'download_verified', 'path': str(path), 'sha256': h.hexdigest(), 'bytes': size}), flush=True) +config = json.loads((ROOT / 'config.json').read_text()) +assert config['architectures'] == ['Qwen3ASRForConditionalGeneration'], config['architectures'] +weights = json.loads((ROOT / 'model.safetensors.index.json').read_text()) +assert set(weights['weight_map'].values()) <= {x['Path'] for x in files} +# Public model weights must remain readable through a read-only bind mount with cap_drop=ALL. +# This does not change the private service .env or acceptance directory permissions. +ROOT.chmod(0o755) +manifest_path.chmod(0o644) +for row in files: + (ROOT / row['Path']).chmod(0o644) +print(json.dumps({'event': 'model_ready', 'model': MODEL, 'revision': REVISION, 'manifest_sha256': digest(manifest_path), 'total_bytes': sum(x['Size'] for x in files), 'architectures': config['architectures'], 'model_type': config.get('model_type'), 'dtype': config.get('torch_dtype', config.get('dtype')), 'weight_metadata': weights['metadata']}), flush=True) diff --git a/deployment/followup-audio-asr/gpu1-mode.py b/deployment/followup-audio-asr/gpu1-mode.py new file mode 100755 index 000000000..003ab047f --- /dev/null +++ b/deployment/followup-audio-asr/gpu1-mode.py @@ -0,0 +1,78 @@ +#!/usr/bin/env python3 +"""Exclusive operational handoff: ASR on GPU1, or GPU1 released for future tests. +Never starts/stops ComfyUI, never kills unknown GPU processes, never prints API keys. +""" +import argparse,fcntl,json,pathlib,subprocess,time,urllib.request +ROOT=pathlib.Path(__file__).resolve().parent +CONTAINER='followup-audio-asr' + +def command(args): + p=subprocess.run(args,capture_output=True,text=True,timeout=700) + if p.returncode: raise RuntimeError('COMMAND_FAILED: '+args[0]) + return p.stdout + +def gpu_pids(): + text=command(['nvidia-smi','--id=1','--query-compute-apps=pid','--format=csv,noheader,nounits']) + return {int(line.strip()) for line in text.splitlines() if line.strip()} + +def container_state(): + p=subprocess.run(['docker','inspect','--format','{{json .State}}',CONTAINER],capture_output=True,text=True) + if p.returncode: + if 'no such object' in p.stderr.lower() or 'no such container' in p.stderr.lower():return {'Running':False,'Status':'absent'} + raise RuntimeError('CONTAINER_STATE_UNKNOWN') + return json.loads(p.stdout) + +def owned_pids(state): + if not state.get('Running'):return set() + lines=command(['docker','top',CONTAINER,'-eo','pid']).splitlines()[1:] + return {int(line.strip()) for line in lines if line.strip()} + +def ensure_exclusive(active,owned): + if active-owned: raise RuntimeError('GPU1_HAS_OTHER_WORKLOAD: refusing to stop or replace it') + +def idle_values(body): + values=[] + for name in ['vllm:num_requests_running','vllm:num_requests_waiting']: + rows=[float(l.rsplit(' ',1)[1]) for l in body.splitlines() if l.startswith(name+'{')] + if not rows:raise RuntimeError('ASR_DRAIN_METRICS_MISSING') + values.append(sum(rows)) + return values + +def drain(): + zeros=0;deadline=time.monotonic()+600 + while time.monotonic()=3:return + time.sleep(5) + raise RuntimeError('ASR_BUSY: unchanged; wait for requests to finish') + +def main(): + p=argparse.ArgumentParser();p.add_argument('mode',choices=['status','asr','test']);args=p.parse_args() + lock=open(ROOT/'.gpu1-mode.lock','a');fcntl.flock(lock,fcntl.LOCK_EX|fcntl.LOCK_NB) + state=container_state();active=gpu_pids();owned=owned_pids(state) + if args.mode=='status': + print(json.dumps({'asr_status':state['Status'],'asr_running':state['Running'],'gpu1_processes':sorted(active),'other_gpu1_processes':sorted(active-owned),'test_service_started':False}));return + ensure_exclusive(active,owned) + compose=['docker','compose','--project-directory',str(ROOT),'--env-file',str(ROOT/'.env'),'-f',str(ROOT/'compose.yaml')] + if args.mode=='test': + if state['Running']: + drain();ensure_exclusive(gpu_pids(),owned_pids(container_state())) + command(compose+['stop','--timeout','660','asr']) + if gpu_pids():raise RuntimeError('GPU1_NOT_FREE') + result={'mode':'test','asr_stopped':True,'gpu1_free':True,'comfyui_test_started':False} + else: + if not state['Running']: + if active:raise RuntimeError('GPU1_NOT_FREE') + command(compose+['up','-d','--no-build','--pull','never','asr']) + deadline=time.monotonic()+480 + while time.monotonic()