62 lines
3.0 KiB
Python
Executable File
62 lines
3.0 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""CPU-only dependency, format and route checks. Does not instantiate a model."""
|
|
import importlib.metadata as metadata
|
|
import io
|
|
import json
|
|
from pathlib import Path
|
|
import numpy as np
|
|
import soundfile as sf
|
|
import soxr
|
|
import scipy
|
|
import av
|
|
import vllm
|
|
import torch
|
|
from transformers import WhisperFeatureExtractor
|
|
|
|
signal=np.zeros(16000,dtype=np.float32)
|
|
encoded=io.BytesIO()
|
|
sf.write(encoded,signal,16000,format='WAV')
|
|
encoded.seek(0)
|
|
decoded,rate=sf.read(encoded,dtype='float32')
|
|
assert rate==16000 and decoded.shape==(16000,)
|
|
resampled=soxr.resample(decoded,16000,8000)
|
|
assert len(resampled)==8000
|
|
features=WhisperFeatureExtractor()(decoded,sampling_rate=rate,return_tensors='np')
|
|
assert features['input_features'].shape[0]==1
|
|
root=Path(vllm.__file__).parent
|
|
model_source=(root/'model_executor/models/qwen3_asr.py').read_text()
|
|
registry_source=(root/'model_executor/models/registry.py').read_text()
|
|
assert 'Qwen3ASRForConditionalGeneration' in registry_source
|
|
assert 'SupportsTranscription' in model_source
|
|
routes=[str(path.relative_to(root)) for path in (root/'entrypoints').rglob('*.py') if '"/v1/audio/transcriptions"' in path.read_text()]
|
|
assert routes, 'Transcription route missing'
|
|
assert not Path('/dev/nvidia0').exists(), 'This test must not have GPU device access'
|
|
assert vllm.__version__=='0.24.0'
|
|
assert metadata.version('numpy')=='2.2.6'
|
|
assert metadata.version('torch')=='2.11.0+cu130'
|
|
assert metadata.version('transformers')=='5.12.1'
|
|
print(json.dumps({'event':'cpu_preflight_pass','versions':{name:metadata.version(name) for name in ['vllm','torch','transformers','numpy','soundfile','scipy','soxr','av']},'wav_roundtrip_shape':list(decoded.shape),'resampled_shape':list(resampled.shape),'feature_shape':list(features['input_features'].shape),'transcription_route_files':routes,'gpu_device_access':False}))
|
|
|
|
# Validate only the installed CLI option declarations. Constructing the full vLLM
|
|
# config on a no-GPU preflight host would try to infer a device.
|
|
import argparse
|
|
import ast
|
|
import inspect
|
|
from vllm.engine.arg_utils import AsyncEngineArgs
|
|
parser = argparse.ArgumentParser()
|
|
tree = ast.parse(Path(inspect.getfile(AsyncEngineArgs)).read_text())
|
|
found = set()
|
|
for node in ast.walk(tree):
|
|
if (isinstance(node, ast.Call) and isinstance(node.func, ast.Attribute)
|
|
and isinstance(node.func.value, ast.Name) and node.func.value.id == "parser"
|
|
and node.func.attr == "add_argument" and node.args
|
|
and isinstance(node.args[0], ast.Constant)
|
|
and node.args[0].value in {"--enable-log-requests", "--shutdown-timeout"}):
|
|
eval(compile(ast.Expression(node), "<installed-option-declaration>", "eval"),
|
|
{"parser": parser, "argparse": argparse, "AsyncEngineArgs": AsyncEngineArgs, "int": int})
|
|
found.add(node.args[0].value)
|
|
assert len(found) == 2
|
|
args = parser.parse_args(["--no-enable-log-requests", "--shutdown-timeout", "600"])
|
|
assert args.enable_log_requests is False and args.shutdown_timeout == 600
|
|
print("CLI_PRIVACY_AND_SHUTDOWN_DECLARATIONS_PASS")
|