[Refactore] [CI] Remove redundant CI test runs step 2 (#17584)
This commit is contained in:
Executable
+89
@@ -0,0 +1,89 @@
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# Detect GPU family from hostname (e.g., linux-mi35x-gpu-1-xxxxx-runner-zzzzz)
|
||||
HOSTNAME_VALUE=$(hostname)
|
||||
GPU_FAMILY=""
|
||||
|
||||
# Host names look like: linux-mi35x-gpu-1-xxxxx-runner-zzzzz
|
||||
if [[ "${HOSTNAME_VALUE}" =~ ^linux-(mi[0-9]+[a-z]*)-gpu-[0-9]+ ]]; then
|
||||
GPU_FAMILY="${BASH_REMATCH[1]}"
|
||||
echo "Detected GPU family from hostname: ${GPU_FAMILY}"
|
||||
else
|
||||
echo "Warning: could not parse GPU family from '${HOSTNAME_VALUE}'"
|
||||
fi
|
||||
|
||||
WORKDIR="/sglang-checkout/test/srt"
|
||||
declare -A ENV_MAP=(
|
||||
[SGLANG_IS_IN_CI_AMD]=1
|
||||
[SGLANG_IS_IN_CI]=1
|
||||
[SGLANG_USE_AITER]=1
|
||||
)
|
||||
|
||||
# Conditionally add GPU_ARCHS only for mi35x
|
||||
if [[ "${GPU_FAMILY}" == "mi35x" ]]; then
|
||||
ENV_MAP[GPU_ARCHS]="gfx950"
|
||||
fi
|
||||
|
||||
# Parse -w/--workdir and -e ENV=VAL
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
-w|--workdir)
|
||||
WORKDIR="$2"
|
||||
shift 2
|
||||
;;
|
||||
-e)
|
||||
IFS="=" read -r key val <<< "$2"
|
||||
ENV_MAP["$key"]="$val"
|
||||
shift 2
|
||||
;;
|
||||
--)
|
||||
shift
|
||||
break
|
||||
;;
|
||||
*)
|
||||
break
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
# Build final ENV_ARGS
|
||||
ENV_ARGS=()
|
||||
for key in "${!ENV_MAP[@]}"; do
|
||||
ENV_ARGS+=("-e" "$key=${ENV_MAP[$key]}")
|
||||
done
|
||||
|
||||
# Run docker exec with retry logic for HuggingFace network/download issues
|
||||
# When HF model downloads fail due to network timeouts or rate limits,
|
||||
# retrying with HF_HUB_OFFLINE=1 uses cached models from previous downloads.
|
||||
#
|
||||
# First attempt: normal mode (allows HF downloads)
|
||||
if docker exec \
|
||||
-w "$WORKDIR" \
|
||||
"${ENV_ARGS[@]}" \
|
||||
ci_sglang "$@"; then
|
||||
exit 0
|
||||
else
|
||||
FIRST_EXIT_CODE=$?
|
||||
fi
|
||||
|
||||
echo "First attempt failed with exit code $FIRST_EXIT_CODE"
|
||||
|
||||
# Skip retry for test failures that won't be fixed by offline mode:
|
||||
# - Exit 1: Test assertion failures (accuracy below threshold)
|
||||
# - Exit 137 (128+9): Process killed by OOM
|
||||
# - Exit 255: Test suite completed with test errors
|
||||
# Only retry for other exit codes (e.g., network timeouts, HF download failures)
|
||||
if [[ "$FIRST_EXIT_CODE" -eq 1 || "$FIRST_EXIT_CODE" -eq 137 || "$FIRST_EXIT_CODE" -eq 255 ]]; then
|
||||
echo "Exit code $FIRST_EXIT_CODE indicates test failure (not network issue), not retrying"
|
||||
exit $FIRST_EXIT_CODE
|
||||
fi
|
||||
|
||||
echo "Retrying with HF_HUB_OFFLINE=1 (offline mode to use cached models)..."
|
||||
|
||||
# Second attempt: force HF offline mode to avoid network timeouts
|
||||
docker exec \
|
||||
-w "$WORKDIR" \
|
||||
"${ENV_ARGS[@]}" \
|
||||
-e HF_HUB_OFFLINE=1 \
|
||||
ci_sglang "$@"
|
||||
Executable
+255
@@ -0,0 +1,255 @@
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
HOSTNAME_VALUE=$(hostname)
|
||||
GPU_ARCH="mi30x" # default
|
||||
OPTIONAL_DEPS="${1:-}"
|
||||
|
||||
# Build python extras
|
||||
EXTRAS="dev_hip"
|
||||
if [ -n "$OPTIONAL_DEPS" ]; then
|
||||
EXTRAS="dev_hip,${OPTIONAL_DEPS}"
|
||||
fi
|
||||
echo "Installing python extras: [${EXTRAS}]"
|
||||
|
||||
# Host names look like: linux-mi35x-gpu-1-xxxxx-runner-zzzzz
|
||||
if [[ "${HOSTNAME_VALUE}" =~ ^linux-(mi[0-9]+[a-z]*)-gpu-[0-9]+ ]]; then
|
||||
GPU_ARCH="${BASH_REMATCH[1]}"
|
||||
echo "Detected GPU architecture from hostname: ${GPU_ARCH}"
|
||||
else
|
||||
echo "Warning: could not parse GPU architecture from '${HOSTNAME_VALUE}', defaulting to ${GPU_ARCH}"
|
||||
fi
|
||||
|
||||
# Install the required dependencies in CI.
|
||||
# Fix permissions on pip cache, ignore errors from concurrent access or missing temp files
|
||||
docker exec ci_sglang chown -R root:root /sgl-data/pip-cache 2>/dev/null || true
|
||||
docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache --upgrade pip
|
||||
docker exec ci_sglang pip uninstall sgl-kernel -y || true
|
||||
docker exec ci_sglang pip uninstall sglang -y || true
|
||||
# Clear Python cache to ensure latest code is used
|
||||
docker exec ci_sglang find /opt/venv -name "*.pyc" -delete || true
|
||||
docker exec ci_sglang find /opt/venv -name "__pycache__" -type d -exec rm -rf {} + || true
|
||||
# Also clear cache in sglang-checkout
|
||||
docker exec ci_sglang find /sglang-checkout -name "*.pyc" -delete || true
|
||||
docker exec ci_sglang find /sglang-checkout -name "__pycache__" -type d -exec rm -rf {} + || true
|
||||
docker exec -w /sglang-checkout/sgl-kernel ci_sglang bash -c "rm -f pyproject.toml && mv pyproject_rocm.toml pyproject.toml && python3 setup_rocm.py install"
|
||||
|
||||
# Helper function to install with retries and fallback PyPI mirror
|
||||
install_with_retry() {
|
||||
local max_attempts=3
|
||||
local cmd="$@"
|
||||
|
||||
for attempt in $(seq 1 $max_attempts); do
|
||||
echo "Attempt $attempt/$max_attempts: $cmd"
|
||||
if eval "$cmd"; then
|
||||
echo "Success!"
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ $attempt -lt $max_attempts ]; then
|
||||
echo "Failed, retrying in 5 seconds..."
|
||||
sleep 5
|
||||
# Try with alternative PyPI index on retry
|
||||
if [[ "$cmd" =~ "pip install" ]] && [ $attempt -eq 2 ]; then
|
||||
cmd="$cmd --index-url https://mirrors.aliyun.com/pypi/simple/ --trusted-host mirrors.aliyun.com"
|
||||
echo "Using fallback PyPI mirror: $cmd"
|
||||
fi
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Failed after $max_attempts attempts"
|
||||
return 1
|
||||
}
|
||||
|
||||
# Helper function to git clone with retries
|
||||
git_clone_with_retry() {
|
||||
local repo_url="$1"
|
||||
local dest_dir="${2:-}"
|
||||
local branch_args="${3:-}"
|
||||
local max_attempts=3
|
||||
|
||||
for attempt in $(seq 1 $max_attempts); do
|
||||
echo "Git clone attempt $attempt/$max_attempts: $repo_url"
|
||||
|
||||
# prevent from partial clone
|
||||
if [ -n "$dest_dir" ] && [ -d "$dest_dir" ]; then
|
||||
rm -rf "$dest_dir"
|
||||
fi
|
||||
|
||||
if git \
|
||||
-c http.lowSpeedLimit=1000 \
|
||||
-c http.lowSpeedTime=30 \
|
||||
clone --depth 1 ${branch_args:+$branch_args} "$repo_url" "$dest_dir"; then
|
||||
echo "Git clone succeeded."
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ $attempt -lt $max_attempts ]; then
|
||||
echo "Git clone failed, retrying in 5 seconds..."
|
||||
sleep 5
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Git clone failed after $max_attempts attempts: $repo_url"
|
||||
return 1
|
||||
}
|
||||
|
||||
|
||||
|
||||
case "${GPU_ARCH}" in
|
||||
mi35x)
|
||||
echo "Runner uses ${GPU_ARCH}; will fetch mi35x image."
|
||||
docker exec ci_sglang rm -rf python/pyproject.toml && mv python/pyproject_other.toml python/pyproject.toml
|
||||
# Follow the same dependency installation flow as mi30x/mi300/mi325.
|
||||
install_with_retry docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e "python[${EXTRAS}]"
|
||||
# For lmms_evals evaluating MMMU
|
||||
docker exec -w / ci_sglang git clone --branch v0.4.1 --depth 1 https://github.com/EvolvingLMMs-Lab/lmms-eval.git
|
||||
install_with_retry docker exec -w /lmms-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e .
|
||||
;;
|
||||
mi30x|mi300|mi325)
|
||||
echo "Runner uses ${GPU_ARCH}; will fetch mi30x image."
|
||||
docker exec ci_sglang rm -rf python/pyproject.toml && mv python/pyproject_other.toml python/pyproject.toml
|
||||
install_with_retry docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e "python[${EXTRAS}]"
|
||||
# For lmms_evals evaluating MMMU
|
||||
docker exec -w / ci_sglang git clone --branch v0.4.1 --depth 1 https://github.com/EvolvingLMMs-Lab/lmms-eval.git
|
||||
install_with_retry docker exec -w /lmms-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e .
|
||||
;;
|
||||
*)
|
||||
echo "Runner architecture '${GPU_ARCH}' unrecognised;" >&2
|
||||
;;
|
||||
esac
|
||||
|
||||
#docker exec -w / ci_sglang git clone https://github.com/merrymercy/human-eval.git
|
||||
git_clone_with_retry https://github.com/merrymercy/human-eval.git human-eval
|
||||
docker cp human-eval ci_sglang:/
|
||||
install_with_retry docker exec -w /human-eval ci_sglang pip install --cache-dir=/sgl-data/pip-cache -e .
|
||||
|
||||
docker exec -w / ci_sglang mkdir -p /dummy-grok
|
||||
mkdir -p dummy-grok && wget https://sharkpublic.blob.core.windows.net/sharkpublic/sglang/dummy_grok.json -O dummy-grok/config.json
|
||||
docker cp ./dummy-grok ci_sglang:/
|
||||
|
||||
docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache huggingface_hub[hf_xet]
|
||||
docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache pytest
|
||||
|
||||
# Install tvm-ffi for JIT kernel support (QK-norm, etc.)
|
||||
echo "Installing tvm-ffi for JIT kernel support..."
|
||||
docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache git+https://github.com/apache/tvm-ffi.git || echo "tvm-ffi installation failed, JIT kernels will use fallback"
|
||||
|
||||
# Install cache-dit for qwen_image_t2i_cache_dit_enabled test (added in PR 16204)
|
||||
docker exec ci_sglang pip install --cache-dir=/sgl-data/pip-cache cache-dit || echo "cache-dit installation failed"
|
||||
|
||||
# Detect AITER version
|
||||
#############################################
|
||||
# Detect correct AITER_COMMIT for this runner
|
||||
# + Check mismatch
|
||||
# + Rebuild AITER if needed
|
||||
#############################################
|
||||
|
||||
echo "[CI-AITER-CHECK] === AITER VERSION CHECK START ==="
|
||||
|
||||
DOCKERFILE="docker/rocm.Dockerfile"
|
||||
|
||||
# GPU_ARCH
|
||||
GPU_ARCH="${GPU_ARCH:-mi30x}"
|
||||
echo "[CI-AITER-CHECK] Runner GPU_ARCH=${GPU_ARCH}"
|
||||
|
||||
#############################################
|
||||
# 1. Extract AITER_COMMIT from correct Dockerfile block
|
||||
#############################################
|
||||
if [[ "${GPU_ARCH}" == "mi35x" ]]; then
|
||||
echo "[CI-AITER-CHECK] Using gfx950 block from Dockerfile..."
|
||||
REPO_AITER_COMMIT=$(grep -F -A20 'FROM $BASE_IMAGE_950 AS gfx950' docker/rocm.Dockerfile \
|
||||
| grep 'AITER_COMMIT=' \
|
||||
| head -n1 \
|
||||
| sed 's/.*AITER_COMMIT="\([^"]*\)".*/\1/')
|
||||
else
|
||||
echo "[CI-AITER-CHECK] Using gfx942-rocm700 block from Dockerfile..."
|
||||
REPO_AITER_COMMIT=$(grep -F -A20 'FROM $BASE_IMAGE_942_ROCM700 AS gfx942-rocm700' docker/rocm.Dockerfile \
|
||||
| grep 'AITER_COMMIT=' \
|
||||
| head -n1 \
|
||||
| sed 's/.*AITER_COMMIT="\([^"]*\)".*/\1/')
|
||||
fi
|
||||
|
||||
|
||||
if [[ -z "${REPO_AITER_COMMIT}" ]]; then
|
||||
echo "[CI-AITER-CHECK] ERROR: Failed to extract AITER_COMMIT from Dockerfile."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "[CI-AITER-CHECK] Dockerfile expects AITER_COMMIT=${REPO_AITER_COMMIT}"
|
||||
|
||||
#############################################
|
||||
# 2. Check container pre-installed AITER version
|
||||
#############################################
|
||||
IMAGE_AITER_VERSION=$(docker exec ci_sglang bash -c "pip show amd-aiter 2>/dev/null | grep '^Version:' | awk '{print \$2}'" || echo "none")
|
||||
IMAGE_AITER_VERSION="v${IMAGE_AITER_VERSION}"
|
||||
echo "[CI-AITER-CHECK] AITER version inside CI image: ${IMAGE_AITER_VERSION}"
|
||||
|
||||
#############################################
|
||||
# 3. Decide rebuild
|
||||
#############################################
|
||||
NEED_REBUILD="false"
|
||||
|
||||
if [[ "${IMAGE_AITER_VERSION}" == "none" ]]; then
|
||||
echo "[CI-AITER-CHECK] No AITER found in image"
|
||||
NEED_REBUILD="true"
|
||||
elif [[ "${IMAGE_AITER_VERSION}" != "${REPO_AITER_COMMIT}" ]]; then
|
||||
echo "[CI-AITER-CHECK] Version mismatch:"
|
||||
echo " Image: ${IMAGE_AITER_VERSION}"
|
||||
echo " Repo : ${REPO_AITER_COMMIT}"
|
||||
NEED_REBUILD="true"
|
||||
else
|
||||
echo "[CI-AITER-CHECK] AITER version matches → using image's version."
|
||||
fi
|
||||
|
||||
|
||||
#############################################
|
||||
# 4. Rebuild AITER if needed
|
||||
#############################################
|
||||
if [[ "${NEED_REBUILD}" == "true" ]]; then
|
||||
echo "[CI-AITER-CHECK] === AITER REBUILD START ==="
|
||||
|
||||
# uninstall existing aiter
|
||||
docker exec ci_sglang pip uninstall -y aiter || true
|
||||
|
||||
# delete old aiter directory
|
||||
docker exec ci_sglang rm -rf /sgl-workspace/aiter
|
||||
|
||||
# clone a fresh copy to /sgl-workspace/aiter
|
||||
docker exec ci_sglang git clone https://github.com/ROCm/aiter.git /sgl-workspace/aiter
|
||||
|
||||
# checkout correct version
|
||||
docker exec ci_sglang bash -c "
|
||||
cd /sgl-workspace/aiter && \
|
||||
git fetch --all && \
|
||||
git checkout ${REPO_AITER_COMMIT} && \
|
||||
git submodule update --init --recursive
|
||||
"
|
||||
|
||||
if [[ "${GPU_ARCH}" == "mi35x" ]]; then
|
||||
GPU_ARCH_LIST="gfx950"
|
||||
else
|
||||
GPU_ARCH_LIST="gfx942"
|
||||
fi
|
||||
echo "[CI-AITER-CHECK] GPU_ARCH_LIST=${GPU_ARCH_LIST}"
|
||||
|
||||
# build AITER
|
||||
docker exec ci_sglang bash -c "
|
||||
cd /sgl-workspace/aiter && \
|
||||
GPU_ARCHS=${GPU_ARCH_LIST} python3 setup.py develop
|
||||
"
|
||||
|
||||
echo "[CI-AITER-CHECK] === AITER REBUILD COMPLETE ==="
|
||||
fi
|
||||
|
||||
echo "[CI-AITER-CHECK] === AITER VERSION CHECK END ==="
|
||||
|
||||
|
||||
# Clear pre-built AITER kernels from Docker image to avoid segfaults
|
||||
# The Docker image may contain pre-compiled kernels incompatible with the current environment
|
||||
echo "Clearing pre-built AITER kernels from Docker image..."
|
||||
docker exec ci_sglang find /sgl-workspace/aiter/aiter/jit -name "*.so" -delete 2>/dev/null || true
|
||||
docker exec ci_sglang ls -la /sgl-workspace/aiter/aiter/jit/ 2>/dev/null || echo "jit dir empty or not found"
|
||||
|
||||
# Pre-build AITER kernels to avoid timeout during tests
|
||||
echo "Warming up AITER JIT kernels..."
|
||||
docker exec -e SGLANG_USE_AITER=1 ci_sglang python3 /sglang-checkout/scripts/ci/amd/amd_ci_warmup_aiter.py || echo "AITER warmup completed (some kernels may not be available)"
|
||||
Executable
+174
@@ -0,0 +1,174 @@
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# Get version from git tags
|
||||
SGLANG_VERSION="v0.5.5" # Default version, will be overridden if git tags are found
|
||||
|
||||
# Fetch tags from origin to ensure we have the latest
|
||||
if git fetch --tags origin; then
|
||||
# Get the latest version tag sorted by version number (e.g., v0.5.7)
|
||||
VERSION_FROM_TAG=$(git tag -l 'v[0-9]*' --sort=-v:refname | head -1)
|
||||
if [ -n "$VERSION_FROM_TAG" ]; then
|
||||
SGLANG_VERSION="$VERSION_FROM_TAG"
|
||||
echo "Using SGLang version from git tags: $SGLANG_VERSION"
|
||||
else
|
||||
echo "Warning: No version tags found; using default $SGLANG_VERSION" >&2
|
||||
fi
|
||||
else
|
||||
echo "Warning: Failed to fetch tags from origin; using default $SGLANG_VERSION" >&2
|
||||
fi
|
||||
|
||||
|
||||
# Default base tags (can be overridden by command line arguments)
|
||||
ROCM_VERSION="rocm700"
|
||||
DEFAULT_MI30X_BASE_TAG="${SGLANG_VERSION}-${ROCM_VERSION}-mi30x"
|
||||
DEFAULT_MI35X_BASE_TAG="${SGLANG_VERSION}-${ROCM_VERSION}-mi35x"
|
||||
|
||||
# Parse command line arguments
|
||||
MI30X_BASE_TAG="${DEFAULT_MI30X_BASE_TAG}"
|
||||
MI35X_BASE_TAG="${DEFAULT_MI35X_BASE_TAG}"
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case $1 in
|
||||
--mi30x-base-tag) MI30X_BASE_TAG="$2"; shift 2;;
|
||||
--mi35x-base-tag) MI35X_BASE_TAG="$2"; shift 2;;
|
||||
-h|--help)
|
||||
echo "Usage: $0 [--mi30x-base-tag TAG] [--mi35x-base-tag TAG]"
|
||||
exit 0
|
||||
;;
|
||||
*) echo "Unknown option $1"; exit 1;;
|
||||
esac
|
||||
done
|
||||
|
||||
|
||||
|
||||
# Detect GPU architecture from the Kubernetes runner hostname
|
||||
HOSTNAME_VALUE=$(hostname)
|
||||
GPU_ARCH="mi30x" # default
|
||||
|
||||
# Host names look like: linux-mi35x-gpu-1-xxxxx-runner-zzzzz
|
||||
if [[ "${HOSTNAME_VALUE}" =~ ^linux-(mi[0-9]+[a-z]*)-gpu-[0-9]+ ]]; then
|
||||
GPU_ARCH="${BASH_REMATCH[1]}"
|
||||
echo "Detected GPU architecture from hostname: ${GPU_ARCH}"
|
||||
else
|
||||
echo "Warning: could not parse GPU architecture from '${HOSTNAME_VALUE}', defaulting to ${GPU_ARCH}"
|
||||
fi
|
||||
|
||||
# Normalise / collapse architectures we don’t yet build specifically for
|
||||
case "${GPU_ARCH}" in
|
||||
mi35x)
|
||||
echo "Runner uses ${GPU_ARCH}; will fetch mi35x image."
|
||||
;;
|
||||
mi30x|mi300|mi325)
|
||||
echo "Runner uses ${GPU_ARCH}; will fetch mi30x image."
|
||||
GPU_ARCH="mi30x"
|
||||
;;
|
||||
*)
|
||||
echo "Runner architecture '${GPU_ARCH}' unrecognised; defaulting to mi30x image." >&2
|
||||
GPU_ARCH="mi30x"
|
||||
;;
|
||||
esac
|
||||
|
||||
|
||||
# Set up DEVICE_FLAG based on Kubernetes pod info
|
||||
if [[ -f /etc/podinfo/gha-render-devices ]]; then
|
||||
DEVICE_FLAG=$(cat /etc/podinfo/gha-render-devices)
|
||||
else
|
||||
DEVICE_FLAG="--device /dev/dri"
|
||||
fi
|
||||
|
||||
|
||||
# Find the latest image
|
||||
find_latest_image() {
|
||||
local gpu_arch=$1
|
||||
local base_tag days_back image_tag
|
||||
|
||||
case "${gpu_arch}" in
|
||||
mi30x) base_tag="${MI30X_BASE_TAG}" ;;
|
||||
mi35x) base_tag="${MI35X_BASE_TAG}" ;;
|
||||
*) echo "Error: unsupported GPU architecture '${gpu_arch}'" >&2; return 1 ;;
|
||||
esac
|
||||
|
||||
# First, check local cache
|
||||
for days_back in {0..6}; do
|
||||
image_tag="${base_tag}-$(date -d "${days_back} days ago" +%Y%m%d)"
|
||||
local local_image="rocm/sgl-dev:${image_tag}"
|
||||
image_id=$(docker images -q "${local_image}")
|
||||
if [[ -n "$image_id" ]]; then
|
||||
echo "Found cached image locally: ${local_image}" >&2
|
||||
echo "${local_image}"
|
||||
return 0
|
||||
fi
|
||||
done
|
||||
|
||||
# If not found locally, fall back to pulling from public registry
|
||||
for days_back in {0..6}; do
|
||||
image_tag="${base_tag}-$(date -d "${days_back} days ago" +%Y%m%d)"
|
||||
echo "Checking for image: rocm/sgl-dev:${image_tag}" >&2
|
||||
if docker manifest inspect "rocm/sgl-dev:${image_tag}" >/dev/null 2>&1; then
|
||||
echo "Found available image: rocm/sgl-dev:${image_tag}" >&2
|
||||
echo "rocm/sgl-dev:${image_tag}"
|
||||
return 0
|
||||
fi
|
||||
done
|
||||
|
||||
# If still not found, try finding any image matching ROCm+arch from remote registry
|
||||
echo "Exact version not found. Searching remote registry for any ${ROCM_VERSION}-${gpu_arch} image…" >&2
|
||||
for days_back in {0..6}; do
|
||||
local target_date=$(date -d "${days_back} days ago" +%Y%m%d)
|
||||
local remote_tags=$(curl -s "https://registry.hub.docker.com/v2/repositories/rocm/sgl-dev/tags?page_size=100&name=${ROCM_VERSION}-${gpu_arch}-${target_date}" 2>/dev/null | grep -o '"name":"[^"]*"' | cut -d'"' -f4 | head -n 1)
|
||||
if [[ -n "$remote_tags" ]]; then
|
||||
echo "Found available image: rocm/sgl-dev:${remote_tags}" >&2
|
||||
echo "rocm/sgl-dev:${remote_tags}"
|
||||
return 0
|
||||
fi
|
||||
done
|
||||
|
||||
echo "No recent images found. Searching any cached local images matching ROCm+arch…" >&2
|
||||
local any_local
|
||||
any_local=$(docker images --format '{{.Repository}}:{{.Tag}}' --filter "reference=rocm/sgl-dev:*${ROCM_VERSION}*${gpu_arch}*" | sort -r | head -n 1)
|
||||
if [[ -n "$any_local" ]]; then
|
||||
echo "Using cached fallback image: ${any_local}" >&2
|
||||
echo "${any_local}"
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo "Error: no ${gpu_arch} image found in the last 7 days for base ${base_tag}" >&2
|
||||
echo "Using hard-coded fallback…" >&2
|
||||
if [[ "${gpu_arch}" == "mi35x" ]]; then
|
||||
echo "rocm/sgl-dev:v0.5.5-rocm700-mi35x-20251110"
|
||||
else
|
||||
echo "rocm/sgl-dev:v0.5.5-rocm700-mi30x-20251110"
|
||||
fi
|
||||
}
|
||||
|
||||
# Pull and run the latest image
|
||||
IMAGE=$(find_latest_image "${GPU_ARCH}")
|
||||
echo "Pulling Docker image: ${IMAGE}"
|
||||
docker pull "${IMAGE}"
|
||||
|
||||
CACHE_HOST=/home/runner/sgl-data
|
||||
if [[ -d "$CACHE_HOST" ]]; then
|
||||
CACHE_VOLUME="-v $CACHE_HOST:/sgl-data"
|
||||
else
|
||||
CACHE_VOLUME=""
|
||||
fi
|
||||
|
||||
echo "Launching container: ci_sglang"
|
||||
docker run -dt --user root --device=/dev/kfd ${DEVICE_FLAG} \
|
||||
-v "${GITHUB_WORKSPACE:-$PWD}:/sglang-checkout" \
|
||||
$CACHE_VOLUME \
|
||||
--group-add video \
|
||||
--shm-size 32g \
|
||||
--cap-add=SYS_PTRACE \
|
||||
-e HF_TOKEN="${HF_TOKEN:-}" \
|
||||
-e HF_HOME=/sgl-data/hf-cache \
|
||||
-e HF_HUB_ETAG_TIMEOUT=300 \
|
||||
-e HF_HUB_DOWNLOAD_TIMEOUT=300 \
|
||||
-e MIOPEN_USER_DB_PATH=/sgl-data/miopen-cache \
|
||||
-e MIOPEN_CUSTOM_CACHE_DIR=/sgl-data/miopen-cache \
|
||||
-e PYTHONPATH="/opt/tilelang:${PYTHONPATH:-}" \
|
||||
--security-opt seccomp=unconfined \
|
||||
-w /sglang-checkout \
|
||||
--name ci_sglang \
|
||||
"${IMAGE}"
|
||||
Executable
+124
@@ -0,0 +1,124 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Warmup script to pre-build AITER JIT kernels.
|
||||
|
||||
This script triggers compilation of commonly used AITER kernels by importing
|
||||
the relevant modules and calling functions with sample data. This avoids
|
||||
timeouts during actual tests when kernels need to be compiled on first use.
|
||||
|
||||
Run this after clearing pre-built AITER kernels from the Docker image.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
# Ensure AITER is enabled
|
||||
os.environ["SGLANG_USE_AITER"] = "1"
|
||||
|
||||
|
||||
def warmup_aiter_kernels():
|
||||
"""Trigger AITER JIT kernel compilation."""
|
||||
import torch
|
||||
|
||||
if not torch.cuda.is_available():
|
||||
print("CUDA/ROCm not available, skipping AITER warmup")
|
||||
return
|
||||
|
||||
print("=" * 60)
|
||||
print("AITER JIT Kernel Warmup")
|
||||
print("=" * 60)
|
||||
|
||||
device = torch.device("cuda:0")
|
||||
start_time = time.time()
|
||||
|
||||
# Warmup RMSNorm kernel (module_rmsnorm) - most commonly used
|
||||
# SGLang uses rmsnorm2d_fwd and rmsnorm2d_fwd_with_add from aiter
|
||||
try:
|
||||
print("\n[1/4] Warming up RMSNorm kernel (rmsnorm2d_fwd)...")
|
||||
from aiter import rmsnorm2d_fwd
|
||||
|
||||
hidden_size = 4096
|
||||
batch_size = 512 # Use larger batch to match CUDA graph capture
|
||||
x = torch.randn(batch_size, hidden_size, dtype=torch.bfloat16, device=device)
|
||||
weight = torch.ones(hidden_size, dtype=torch.bfloat16, device=device)
|
||||
eps = 1e-6
|
||||
|
||||
# This triggers JIT compilation
|
||||
_ = rmsnorm2d_fwd(x, weight, eps)
|
||||
torch.cuda.synchronize()
|
||||
print(f" RMSNorm kernel (rmsnorm2d_fwd) compiled successfully")
|
||||
except Exception as e:
|
||||
print(f" RMSNorm warmup failed (may not be available): {e}")
|
||||
|
||||
# Warmup fused add RMSNorm kernel
|
||||
try:
|
||||
print("\n[2/4] Warming up fused add RMSNorm kernel (rmsnorm2d_fwd_with_add)...")
|
||||
from aiter import rmsnorm2d_fwd_with_add
|
||||
|
||||
hidden_size = 4096
|
||||
batch_size = 512
|
||||
x = torch.randn(batch_size, hidden_size, dtype=torch.bfloat16, device=device)
|
||||
residual = torch.randn(
|
||||
batch_size, hidden_size, dtype=torch.bfloat16, device=device
|
||||
)
|
||||
weight = torch.ones(hidden_size, dtype=torch.bfloat16, device=device)
|
||||
eps = 1e-6
|
||||
|
||||
# This triggers JIT compilation
|
||||
_ = rmsnorm2d_fwd_with_add(x, residual, weight, eps)
|
||||
torch.cuda.synchronize()
|
||||
print(f" Fused add RMSNorm kernel compiled successfully")
|
||||
except Exception as e:
|
||||
print(f" Fused add RMSNorm warmup failed (may not be available): {e}")
|
||||
|
||||
# Warmup rotary embedding kernel if available
|
||||
try:
|
||||
print("\n[3/4] Warming up rotary embedding kernel...")
|
||||
from aiter import rotary_embedding
|
||||
|
||||
head_size = 128
|
||||
seq_len = 32
|
||||
num_heads = 32
|
||||
positions = torch.arange(seq_len, device=device)
|
||||
query = torch.randn(
|
||||
seq_len, num_heads, head_size, dtype=torch.bfloat16, device=device
|
||||
)
|
||||
key = torch.randn(
|
||||
seq_len, num_heads, head_size, dtype=torch.bfloat16, device=device
|
||||
)
|
||||
cos = torch.ones(seq_len, head_size // 2, dtype=torch.bfloat16, device=device)
|
||||
sin = torch.zeros(seq_len, head_size // 2, dtype=torch.bfloat16, device=device)
|
||||
|
||||
_ = rotary_embedding(positions, query, key, head_size, cos, sin, True)
|
||||
torch.cuda.synchronize()
|
||||
print(f" Rotary embedding kernel compiled successfully")
|
||||
except Exception as e:
|
||||
print(f" Rotary embedding warmup skipped (may not be available): {e}")
|
||||
|
||||
# Warmup activation kernels if available
|
||||
try:
|
||||
print("\n[4/4] Warming up activation kernels...")
|
||||
from aiter import silu_and_mul
|
||||
|
||||
hidden_size = 4096
|
||||
batch_size = 512
|
||||
x = torch.randn(
|
||||
batch_size, hidden_size * 2, dtype=torch.bfloat16, device=device
|
||||
)
|
||||
out = torch.empty(batch_size, hidden_size, dtype=torch.bfloat16, device=device)
|
||||
|
||||
silu_and_mul(out, x)
|
||||
torch.cuda.synchronize()
|
||||
print(f" Activation kernel compiled successfully")
|
||||
except Exception as e:
|
||||
print(f" Activation warmup skipped (may not be available): {e}")
|
||||
|
||||
elapsed = time.time() - start_time
|
||||
print("\n" + "=" * 60)
|
||||
print(f"AITER warmup completed in {elapsed:.1f}s")
|
||||
print("=" * 60 + "\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
warmup_aiter_kernels()
|
||||
Executable
+60
@@ -0,0 +1,60 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Simple RCCL test for multi-GPU communication.
|
||||
This test verifies that RCCL can initialize and communicate across multiple GPUs.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
import torch
|
||||
import torch.distributed as dist
|
||||
|
||||
|
||||
def test_rccl_allreduce():
|
||||
"""Test basic RCCL allreduce operation across all GPUs."""
|
||||
if not torch.cuda.is_available():
|
||||
print("CUDA not available, skipping test")
|
||||
sys.exit(1)
|
||||
|
||||
# Initialize process group with NCCL (RCCL on AMD)
|
||||
dist.init_process_group(backend="nccl")
|
||||
|
||||
rank = dist.get_rank()
|
||||
world_size = dist.get_world_size()
|
||||
|
||||
print(f"[Rank {rank}/{world_size}] Initialized successfully")
|
||||
|
||||
# Set device
|
||||
device = torch.device(f"cuda:{rank}")
|
||||
torch.cuda.set_device(device)
|
||||
|
||||
print(f"[Rank {rank}] Device: {torch.cuda.get_device_name(device)}")
|
||||
print(
|
||||
f"[Rank {rank}] Device memory: {torch.cuda.get_device_properties(device).total_memory / 1e9:.2f} GB"
|
||||
)
|
||||
|
||||
# Create a tensor and perform allreduce
|
||||
tensor = torch.ones(1000, device=device) * rank
|
||||
print(f"[Rank {rank}] Before allreduce: tensor sum = {tensor.sum().item()}")
|
||||
|
||||
dist.all_reduce(tensor, op=dist.ReduceOp.SUM)
|
||||
|
||||
expected_sum = sum(range(world_size)) * 1000
|
||||
actual_sum = tensor.sum().item()
|
||||
|
||||
print(
|
||||
f"[Rank {rank}] After allreduce: tensor sum = {actual_sum}, expected = {expected_sum}"
|
||||
)
|
||||
|
||||
if abs(actual_sum - expected_sum) < 0.1:
|
||||
print(f"[Rank {rank}] ✓ RCCL allreduce test PASSED")
|
||||
dist.destroy_process_group()
|
||||
sys.exit(0)
|
||||
else:
|
||||
print(f"[Rank {rank}] ✗ RCCL allreduce test FAILED")
|
||||
dist.destroy_process_group()
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
test_rccl_allreduce()
|
||||
Reference in New Issue
Block a user