name: macOS Apple Silicon Smoke Test on: schedule: # Daily at 2:30 AM UTC - cron: '30 2 * * *' workflow_dispatch: # Manual trigger permissions: contents: read jobs: macos-m1-smoke-test: # macos-26 (the supported target) is still a preview runner, so gate on GA # macos-15 and keep macos-26 non-blocking. strategy: fail-fast: false matrix: include: - os: macos-15 required: true - os: macos-26 required: false name: macos-m1-smoke-test (${{ matrix.os }}) runs-on: ${{ matrix.os }} continue-on-error: ${{ !matrix.required }} timeout-minutes: 30 steps: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 - uses: astral-sh/setup-uv@37802adc94f370d6bfd71619e3f0bf239e1f3b78 # v7.6.0 with: enable-cache: true cache-dependency-glob: | requirements/**/*.txt pyproject.toml python-version: '3.12' - name: Create virtual environment run: | uv venv echo "$GITHUB_WORKSPACE/.venv/bin" >> "$GITHUB_PATH" - name: Install dependencies and build vLLM run: | uv pip install -r requirements/build/cpu.txt --index-strategy unsafe-best-match uv pip install -r requirements/cpu.txt --index-strategy unsafe-best-match uv pip install -e . --no-build-isolation env: CMAKE_BUILD_PARALLEL_LEVEL: 4 - name: Verify installation run: | python -c "import vllm; print(f'vLLM version: {vllm.__version__}')" - name: Smoke test vllm serve run: | # Start server in background VLLM_CPU_KVCACHE_SPACE=1 \ vllm serve Qwen/Qwen3-0.6B \ --max-model-len=2K \ --load-format=dummy \ --hf-overrides '{"num_hidden_layers": 2}' \ --enforce-eager \ --port 8000 & SERVER_PID=$! # Wait for server to start for i in {1..30}; do if curl -s http://localhost:8000/health > /dev/null; then echo "Server started successfully" break fi if [ "$i" -eq 30 ]; then echo "Server failed to start" kill "$SERVER_PID" exit 1 fi sleep 2 done # Test health endpoint curl -f http://localhost:8000/health # Long prompt: hits the split-KV path that short prompts skip (#46769). PAYLOAD=$(python -c "import json; print(json.dumps({'model': 'Qwen/Qwen3-0.6B', 'prompt': 'The quick brown fox jumps over the lazy dog. ' * 24, 'max_tokens': 16}))") curl -f --max-time 120 http://localhost:8000/v1/completions \ -H "Content-Type: application/json" \ -d "$PAYLOAD" # Cleanup kill "$SERVER_PID"