#!/bin/bash set -euo pipefail HOSTNAME_VALUE=$(hostname) GPU_ARCH="mi30x" # default # Host names look like: linux-mi35x-gpu-1-xxxxx-runner-zzzzz if [[ "${HOSTNAME_VALUE}" =~ ^linux-(mi[0-9]+[a-z]*)-gpu-[0-9]+ ]]; then GPU_ARCH="${BASH_REMATCH[1]}" echo "Detected GPU architecture from hostname: ${GPU_ARCH}" else echo "Warning: could not parse GPU architecture from '${HOSTNAME_VALUE}', defaulting to ${GPU_ARCH}" fi # Install the required dependencies in CI. # Fix permissions on pip cache, ignore errors from concurrent access or missing temp files docker exec ci_sglang chown -R root:root /sgl-data/pip-cache 2>/dev/null || true docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache --upgrade pip docker exec ci_sglang pip uninstall sgl-kernel -y || true # Clear Python cache to ensure latest code is used docker exec ci_sglang find /opt/venv -name "*.pyc" -delete || true docker exec ci_sglang find /opt/venv -name "__pycache__" -type d -exec rm -rf {} + || true docker exec -w /sglang-checkout/sgl-kernel ci_sglang bash -c "rm -f pyproject.toml && mv pyproject_rocm.toml pyproject.toml && python3 setup_rocm.py install" # Helper function to install with retries and fallback PyPI mirror install_with_retry() { local max_attempts=3 local cmd="$@" for attempt in $(seq 1 $max_attempts); do echo "Attempt $attempt/$max_attempts: $cmd" if eval "$cmd"; then echo "Success!" return 0 fi if [ $attempt -lt $max_attempts ]; then echo "Failed, retrying in 5 seconds..." sleep 5 # Try with alternative PyPI index on retry if [[ "$cmd" =~ "pip install" ]] && [ $attempt -eq 2 ]; then cmd="$cmd --index-url https://mirrors.aliyun.com/pypi/simple/ --trusted-host mirrors.aliyun.com" echo "Using fallback PyPI mirror: $cmd" fi fi done echo "Failed after $max_attempts attempts" return 1 } case "${GPU_ARCH}" in mi35x) echo "Runner uses ${GPU_ARCH}; will fetch mi35x image." docker exec ci_sglang rm -rf python/pyproject.toml && mv python/pyproject_other.toml python/pyproject.toml install_with_retry docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e "python[dev_hip]" --no-deps # For lmms_evals evaluating MMMU docker exec -w / ci_sglang git clone --branch v0.4.1 --depth 1 https://github.com/EvolvingLMMs-Lab/lmms-eval.git install_with_retry docker exec -w /lmms-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e . --no-deps ;; mi30x|mi300|mi325) echo "Runner uses ${GPU_ARCH}; will fetch mi30x image." docker exec ci_sglang rm -rf python/pyproject.toml && mv python/pyproject_other.toml python/pyproject.toml install_with_retry docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e "python[dev_hip]" # For lmms_evals evaluating MMMU docker exec -w / ci_sglang git clone --branch v0.4.1 --depth 1 https://github.com/EvolvingLMMs-Lab/lmms-eval.git install_with_retry docker exec -w /lmms-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e . ;; *) echo "Runner architecture '${GPU_ARCH}' unrecognised;" >&2 ;; esac docker exec -w / ci_sglang git clone https://github.com/merrymercy/human-eval.git install_with_retry docker exec -w /human-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e . docker exec -w / ci_sglang mkdir -p /dummy-grok mkdir -p dummy-grok && wget https://sharkpublic.blob.core.windows.net/sharkpublic/sglang/dummy_grok.json -O dummy-grok/config.json docker cp ./dummy-grok ci_sglang:/ docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache huggingface_hub[hf_xet] docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache pytest