ops: add isolated ASR deployment and safe GPU test handoff

This commit is contained in:
2026-10-08 12:42:01 +08:00
parent fea33285e8
commit 8bb4bd07ae
14 changed files with 530 additions and 0 deletions
+61
View File
@@ -0,0 +1,61 @@
#!/usr/bin/env python3
"""CPU-only dependency, format and route checks. Does not instantiate a model."""
import importlib.metadata as metadata
import io
import json
from pathlib import Path
import numpy as np
import soundfile as sf
import soxr
import scipy
import av
import vllm
import torch
from transformers import WhisperFeatureExtractor
signal=np.zeros(16000,dtype=np.float32)
encoded=io.BytesIO()
sf.write(encoded,signal,16000,format='WAV')
encoded.seek(0)
decoded,rate=sf.read(encoded,dtype='float32')
assert rate==16000 and decoded.shape==(16000,)
resampled=soxr.resample(decoded,16000,8000)
assert len(resampled)==8000
features=WhisperFeatureExtractor()(decoded,sampling_rate=rate,return_tensors='np')
assert features['input_features'].shape[0]==1
root=Path(vllm.__file__).parent
model_source=(root/'model_executor/models/qwen3_asr.py').read_text()
registry_source=(root/'model_executor/models/registry.py').read_text()
assert 'Qwen3ASRForConditionalGeneration' in registry_source
assert 'SupportsTranscription' in model_source
routes=[str(path.relative_to(root)) for path in (root/'entrypoints').rglob('*.py') if '"/v1/audio/transcriptions"' in path.read_text()]
assert routes, 'Transcription route missing'
assert not Path('/dev/nvidia0').exists(), 'This test must not have GPU device access'
assert vllm.__version__=='0.24.0'
assert metadata.version('numpy')=='2.2.6'
assert metadata.version('torch')=='2.11.0+cu130'
assert metadata.version('transformers')=='5.12.1'
print(json.dumps({'event':'cpu_preflight_pass','versions':{name:metadata.version(name) for name in ['vllm','torch','transformers','numpy','soundfile','scipy','soxr','av']},'wav_roundtrip_shape':list(decoded.shape),'resampled_shape':list(resampled.shape),'feature_shape':list(features['input_features'].shape),'transcription_route_files':routes,'gpu_device_access':False}))
# Validate only the installed CLI option declarations. Constructing the full vLLM
# config on a no-GPU preflight host would try to infer a device.
import argparse
import ast
import inspect
from vllm.engine.arg_utils import AsyncEngineArgs
parser = argparse.ArgumentParser()
tree = ast.parse(Path(inspect.getfile(AsyncEngineArgs)).read_text())
found = set()
for node in ast.walk(tree):
if (isinstance(node, ast.Call) and isinstance(node.func, ast.Attribute)
and isinstance(node.func.value, ast.Name) and node.func.value.id == "parser"
and node.func.attr == "add_argument" and node.args
and isinstance(node.args[0], ast.Constant)
and node.args[0].value in {"--enable-log-requests", "--shutdown-timeout"}):
eval(compile(ast.Expression(node), "<installed-option-declaration>", "eval"),
{"parser": parser, "argparse": argparse, "AsyncEngineArgs": AsyncEngineArgs, "int": int})
found.add(node.args[0].value)
assert len(found) == 2
args = parser.parse_args(["--no-enable-log-requests", "--shutdown-timeout", "600"])
assert args.enable_log_requests is False and args.shutdown_timeout == 600
print("CLI_PRIVACY_AND_SHUTDOWN_DECLARATIONS_PASS")