#!/bin/bash set -euo pipefail HOSTNAME_VALUE=$(hostname) GPU_ARCH="mi30x" # default OPTIONAL_DEPS="${1:-}" # Build python extras EXTRAS="dev_hip" if [ -n "$OPTIONAL_DEPS" ]; then EXTRAS="dev_hip,${OPTIONAL_DEPS}" fi echo "Installing python extras: [${EXTRAS}]" # Host names look like: linux-mi35x-gpu-1-xxxxx-runner-zzzzz if [[ "${HOSTNAME_VALUE}" =~ ^linux-(mi[0-9]+[a-z]*)-gpu-[0-9]+ ]]; then GPU_ARCH="${BASH_REMATCH[1]}" echo "Detected GPU architecture from hostname: ${GPU_ARCH}" else echo "Warning: could not parse GPU architecture from '${HOSTNAME_VALUE}', defaulting to ${GPU_ARCH}" fi # Install the required dependencies in CI. # Fix permissions on pip cache, ignore errors from concurrent access or missing temp files docker exec ci_sglang chown -R root:root /sgl-data/pip-cache 2>/dev/null || true docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache --upgrade pip docker exec ci_sglang pip uninstall sgl-kernel -y || true docker exec ci_sglang pip uninstall sglang -y || true # Clear Python cache to ensure latest code is used docker exec ci_sglang find /opt/venv -name "*.pyc" -delete || true docker exec ci_sglang find /opt/venv -name "__pycache__" -type d -exec rm -rf {} + || true # Also clear cache in sglang-checkout docker exec ci_sglang find /sglang-checkout -name "*.pyc" -delete || true docker exec ci_sglang find /sglang-checkout -name "__pycache__" -type d -exec rm -rf {} + || true docker exec -w /sglang-checkout/sgl-kernel ci_sglang bash -c "rm -f pyproject.toml && mv pyproject_rocm.toml pyproject.toml && python3 setup_rocm.py install" # Helper function to install with retries and fallback PyPI mirror install_with_retry() { local max_attempts=3 local cmd="$@" for attempt in $(seq 1 $max_attempts); do echo "Attempt $attempt/$max_attempts: $cmd" if eval "$cmd"; then echo "Success!" return 0 fi if [ $attempt -lt $max_attempts ]; then echo "Failed, retrying in 5 seconds..." sleep 5 # Try with alternative PyPI index on retry if [[ "$cmd" =~ "pip install" ]] && [ $attempt -eq 2 ]; then cmd="$cmd --index-url https://mirrors.aliyun.com/pypi/simple/ --trusted-host mirrors.aliyun.com" echo "Using fallback PyPI mirror: $cmd" fi fi done echo "Failed after $max_attempts attempts" return 1 } case "${GPU_ARCH}" in mi35x) echo "Runner uses ${GPU_ARCH}; will fetch mi35x image." docker exec ci_sglang rm -rf python/pyproject.toml && mv python/pyproject_other.toml python/pyproject.toml install_with_retry docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e "python[${EXTRAS}]" --no-deps # TODO: only for mi35x # For lmms_evals evaluating MMMU docker exec -w / ci_sglang git clone --branch v0.4.1 --depth 1 https://github.com/EvolvingLMMs-Lab/lmms-eval.git install_with_retry docker exec -w /lmms-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e . --no-deps ;; mi30x|mi300|mi325) echo "Runner uses ${GPU_ARCH}; will fetch mi30x image." docker exec ci_sglang rm -rf python/pyproject.toml && mv python/pyproject_other.toml python/pyproject.toml install_with_retry docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e "python[${EXTRAS}]" # For lmms_evals evaluating MMMU docker exec -w / ci_sglang git clone --branch v0.4.1 --depth 1 https://github.com/EvolvingLMMs-Lab/lmms-eval.git install_with_retry docker exec -w /lmms-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e . ;; *) echo "Runner architecture '${GPU_ARCH}' unrecognised;" >&2 ;; esac docker exec -w / ci_sglang git clone https://github.com/merrymercy/human-eval.git install_with_retry docker exec -w /human-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e . docker exec -w / ci_sglang mkdir -p /dummy-grok mkdir -p dummy-grok && wget https://sharkpublic.blob.core.windows.net/sharkpublic/sglang/dummy_grok.json -O dummy-grok/config.json docker cp ./dummy-grok ci_sglang:/ docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache huggingface_hub[hf_xet] docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache pytest # Clear pre-built AITER kernels from Docker image to avoid segfaults # The Docker image may contain pre-compiled kernels incompatible with the current environment echo "Clearing pre-built AITER kernels from Docker image..." docker exec ci_sglang find /sgl-workspace/aiter/aiter/jit -name "*.so" -delete 2>/dev/null || true docker exec ci_sglang ls -la /sgl-workspace/aiter/aiter/jit/ 2>/dev/null || echo "jit dir empty or not found" # Pre-build AITER kernels to avoid timeout during tests echo "Warming up AITER JIT kernels..." docker exec -e SGLANG_USE_AITER=1 ci_sglang python3 /sglang-checkout/scripts/ci/amd_ci_warmup_aiter.py || echo "AITER warmup completed (some kernels may not be available)"