ops: add isolated ASR deployment and safe GPU test handoff
This commit is contained in:
+61
@@ -0,0 +1,61 @@
|
||||
#!/usr/bin/env python3
|
||||
"""CPU-only dependency, format and route checks. Does not instantiate a model."""
|
||||
import importlib.metadata as metadata
|
||||
import io
|
||||
import json
|
||||
from pathlib import Path
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
import soxr
|
||||
import scipy
|
||||
import av
|
||||
import vllm
|
||||
import torch
|
||||
from transformers import WhisperFeatureExtractor
|
||||
|
||||
signal=np.zeros(16000,dtype=np.float32)
|
||||
encoded=io.BytesIO()
|
||||
sf.write(encoded,signal,16000,format='WAV')
|
||||
encoded.seek(0)
|
||||
decoded,rate=sf.read(encoded,dtype='float32')
|
||||
assert rate==16000 and decoded.shape==(16000,)
|
||||
resampled=soxr.resample(decoded,16000,8000)
|
||||
assert len(resampled)==8000
|
||||
features=WhisperFeatureExtractor()(decoded,sampling_rate=rate,return_tensors='np')
|
||||
assert features['input_features'].shape[0]==1
|
||||
root=Path(vllm.__file__).parent
|
||||
model_source=(root/'model_executor/models/qwen3_asr.py').read_text()
|
||||
registry_source=(root/'model_executor/models/registry.py').read_text()
|
||||
assert 'Qwen3ASRForConditionalGeneration' in registry_source
|
||||
assert 'SupportsTranscription' in model_source
|
||||
routes=[str(path.relative_to(root)) for path in (root/'entrypoints').rglob('*.py') if '"/v1/audio/transcriptions"' in path.read_text()]
|
||||
assert routes, 'Transcription route missing'
|
||||
assert not Path('/dev/nvidia0').exists(), 'This test must not have GPU device access'
|
||||
assert vllm.__version__=='0.24.0'
|
||||
assert metadata.version('numpy')=='2.2.6'
|
||||
assert metadata.version('torch')=='2.11.0+cu130'
|
||||
assert metadata.version('transformers')=='5.12.1'
|
||||
print(json.dumps({'event':'cpu_preflight_pass','versions':{name:metadata.version(name) for name in ['vllm','torch','transformers','numpy','soundfile','scipy','soxr','av']},'wav_roundtrip_shape':list(decoded.shape),'resampled_shape':list(resampled.shape),'feature_shape':list(features['input_features'].shape),'transcription_route_files':routes,'gpu_device_access':False}))
|
||||
|
||||
# Validate only the installed CLI option declarations. Constructing the full vLLM
|
||||
# config on a no-GPU preflight host would try to infer a device.
|
||||
import argparse
|
||||
import ast
|
||||
import inspect
|
||||
from vllm.engine.arg_utils import AsyncEngineArgs
|
||||
parser = argparse.ArgumentParser()
|
||||
tree = ast.parse(Path(inspect.getfile(AsyncEngineArgs)).read_text())
|
||||
found = set()
|
||||
for node in ast.walk(tree):
|
||||
if (isinstance(node, ast.Call) and isinstance(node.func, ast.Attribute)
|
||||
and isinstance(node.func.value, ast.Name) and node.func.value.id == "parser"
|
||||
and node.func.attr == "add_argument" and node.args
|
||||
and isinstance(node.args[0], ast.Constant)
|
||||
and node.args[0].value in {"--enable-log-requests", "--shutdown-timeout"}):
|
||||
eval(compile(ast.Expression(node), "<installed-option-declaration>", "eval"),
|
||||
{"parser": parser, "argparse": argparse, "AsyncEngineArgs": AsyncEngineArgs, "int": int})
|
||||
found.add(node.args[0].value)
|
||||
assert len(found) == 2
|
||||
args = parser.parse_args(["--no-enable-log-requests", "--shutdown-timeout", "600"])
|
||||
assert args.enable_log_requests is False and args.shutdown_timeout == 600
|
||||
print("CLI_PRIVACY_AND_SHUTDOWN_DECLARATIONS_PASS")
|
||||
Reference in New Issue
Block a user