chore: import upstream snapshot with attribution
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
PR Test (NPU) / check-changes (push) Has been cancelled
PR Test (NPU) / pr-gate (push) Has been cancelled
PR Test (NPU) / set-image-config (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-1-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (0) (push) Has been cancelled
PR Test (NPU) / stage-b-test-2-npu-a2 (1) (push) Has been cancelled
PR Test (NPU) / stage-b-test-4-npu-a3 (push) Has been cancelled
PR Test (NPU) / stage-b-test-16-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-1-npu-a3 (push) Has been cancelled
PR Test (NPU) / multimodal-gen-test-2-npu-a3 (push) Has been cancelled
PR Test (Arm64) / pr-gate (push) Has been cancelled
PR Test (Arm64) / check-changes (push) Has been cancelled
PR Test (Arm64) / build-test (push) Has been cancelled
PR Test (sgl-router) / gate (push) Has been cancelled
PR Test (sgl-router) / tier-1 — lint (push) Has been cancelled
PR Test (sgl-router) / tier-2 — build + test (push) Has been cancelled
PR Test (sgl-router) / tier-3 — docker (placeholder) (push) Has been cancelled
PR Test (sgl-router) / tier-3 — k8s integration (push) Has been cancelled
PR Test (sgl-router) / tier-3 — e2e (push) Has been cancelled
PR Test (sgl-router) / finish (push) Has been cancelled
PR Test (NPU) / single-node-poc (map[name:qwen3_6_27b_w8a8_1p_in64k_out1k_50ms runner:linux-aarch64-a3-2 test_case:test/registered/ascend/performance/qwen3_6_27b/test_npu_qwen3_6_27b_w8a8_1p_in64k_out1k_50ms.py test_type:perf]) (push) Has been cancelled
PR Test (NPU) / pr-test-npu-finish (push) Has been cancelled
PR Test (Xeon) / pr-gate (push) Has been cancelled
PR Test (Xeon) / check-changes (push) Has been cancelled
PR Test (Xeon) / build-test (, xeon-gnr, base-b-test-cpu) (push) Has been cancelled
PR Test (XPU) / check-changes (push) Has been cancelled
PR Test (XPU) / pr-gate (push) Has been cancelled
PR Test (XPU) / stage-a-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / wait-for-stage-a (push) Has been cancelled
PR Test (XPU) / stage-b-test-1-gpu-xpu (push) Has been cancelled
PR Test (XPU) / finish (push) Has been cancelled
CI Model Inventory / build-inventory (push) Has been cancelled
Lint / lint (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Compilation Check (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Manual Policy (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark - Request Processing (push) Has been cancelled
PR Benchmark (SMG Components) / Benchmark Summary (push) Has been cancelled
PR Test (SMG) / build-wheel (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on windows (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (x86_64 - auto) (push) Has been cancelled
PR Test (SMG) / python-unit-tests (push) Has been cancelled
PR Test (SMG) / unit-tests (push) Has been cancelled
PR Test (SMG) / benchmarks (push) Has been cancelled
PR Test (SMG) / chat-completions (push) Has been cancelled
PR Test (SMG) / chat-completions-4gpu (push) Has been cancelled
PR Test (SMG) / e2e (push) Has been cancelled
PR Test (SMG) / docker-build-test (push) Has been cancelled
PR Test (SMG) / k8s-integration (push) Has been cancelled
PR Test (SMG) / finish (push) Has been cancelled
PR Test (SMG) / summarize-benchmarks (push) Has been cancelled
Release SGLang Model Gateway Docker Image / publish (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on macos (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - auto) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (aarch64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / build on linux (x86_64 - musllinux_1_1) (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Build SDist (push) Has been cancelled
Release SGLang Model Gateway to PyPI / Upload to PyPI (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (aarch64, 12.9, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu129-matrix (x86_64, 12.9, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu129 (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (aarch64, 13.0, 3.10, arm-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / build-cu130-matrix (x86_64, 13.0, 3.10, x64-kernel-build-node) (push) Has been cancelled
Release SGLang Kernels / release-cu130 (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 700) (push) Has been cancelled
Release SGLang Kernels / build-rocm-matrix (3.10, 720) (push) Has been cancelled
Release SGLang Kernels / release-rocm700 (push) Has been cancelled
Release SGLang Kernels / release-rocm720 (push) Has been cancelled
Release SGLang Kernels / build-musa43 (43, 3.10) (push) Has been cancelled
Release SGLang Kernels / release-musa43 (push) Has been cancelled
This commit is contained in:
@@ -0,0 +1,897 @@
|
||||
ARG CUDA_VERSION=13.0.1
|
||||
FROM nvidia/cuda:${CUDA_VERSION}-cudnn-devel-ubuntu24.04 AS base
|
||||
|
||||
ARG TARGETARCH
|
||||
ARG BUILD_TYPE=all
|
||||
ARG BRANCH_TYPE=remote
|
||||
ARG GRACE_BLACKWELL=0
|
||||
ARG HOPPER_SBO=0
|
||||
|
||||
ARG HOPPER_SBO_DEEPEP_COMMIT=9f2fc4b3182a51044ae7ecb6610f7c9c3258c4d6
|
||||
ARG DEEPEP_COMMIT=9af0e0d0e74f3577af1979c9b9e1ac2cad0104ee
|
||||
ARG BUILD_AND_DOWNLOAD_PARALLEL=8
|
||||
ARG SGL_KERNEL_VERSION=0.4.4
|
||||
ARG SGL_VERSION
|
||||
ARG SGL_DEEP_GEMM_VERSION=0.1.4.post1
|
||||
ARG USE_LATEST_SGLANG=0
|
||||
ARG GDRCOPY_VERSION=2.5.1
|
||||
ARG PIP_DEFAULT_INDEX
|
||||
ARG UBUNTU_MIRROR
|
||||
ARG GITHUB_ARTIFACTORY=github.com
|
||||
ARG INSTALL_FLASHINFER_JIT_CACHE=0
|
||||
ARG FLASHINFER_VERSION=0.6.14
|
||||
ARG MOONCAKE_VERSION=0.3.11.post1
|
||||
ARG MSCCLPP_VERSION=sglang-v0.9.1
|
||||
#if need other arg please add in MOONCAKE_COMPILE_ARG
|
||||
ARG MOONCAKE_COMPILE_ARG="-DUSE_HTTP=ON -DUSE_MNNVL=ON -DUSE_CUDA=ON -DWITH_EP=ON"
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive \
|
||||
CUDA_HOME=/usr/local/cuda \
|
||||
GDRCOPY_HOME=/usr/src/gdrdrv-${GDRCOPY_VERSION}/ \
|
||||
FLASHINFER_VERSION=${FLASHINFER_VERSION}
|
||||
|
||||
# Add GKE default lib and bin locations
|
||||
ENV PATH="${PATH}:/usr/local/nvidia/bin" \
|
||||
LD_LIBRARY_PATH="${LD_LIBRARY_PATH}:/usr/local/nvidia/lib:/usr/local/nvidia/lib64"
|
||||
|
||||
# Replace Ubuntu sources if specified
|
||||
RUN if [ -n "$UBUNTU_MIRROR" ]; then \
|
||||
sed -i "s|http://.*archive.ubuntu.com|$UBUNTU_MIRROR|g" /etc/apt/sources.list && \
|
||||
sed -i "s|http://.*security.ubuntu.com|$UBUNTU_MIRROR|g" /etc/apt/sources.list; \
|
||||
fi
|
||||
|
||||
# Python setup (combined with apt update to reduce layers)
|
||||
# Ubuntu 24.04 ships Python 3.12 in main, so we no longer need the deadsnakes
|
||||
# PPA. Dropping it avoids transient Launchpad 504s in `add-apt-repository`.
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=base-apt \
|
||||
apt update && apt install -y --no-install-recommends wget software-properties-common \
|
||||
&& apt install -y --no-install-recommends python3.12-full python3.12-dev \
|
||||
&& update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.12 2 \
|
||||
&& update-alternatives --set python3 /usr/bin/python3.12 \
|
||||
&& wget -q https://bootstrap.pypa.io/get-pip.py \
|
||||
&& python3 get-pip.py --break-system-packages \
|
||||
&& rm get-pip.py \
|
||||
# Allow pip to install packages globally (PEP 668 workaround for Ubuntu 24.04)
|
||||
&& python3 -m pip config set global.break-system-packages true \
|
||||
# Fix for apt-add-repository
|
||||
&& cd /usr/lib/python3/dist-packages/ \
|
||||
&& ln -s apt_pkg.cpython-312-*-linux-gnu.so apt_pkg.so
|
||||
|
||||
# Install system dependencies (organized by category for better caching)
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=base-apt \
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
# Core system utilities
|
||||
ca-certificates \
|
||||
software-properties-common \
|
||||
netcat-openbsd \
|
||||
kmod \
|
||||
unzip \
|
||||
openssh-server \
|
||||
curl \
|
||||
wget \
|
||||
lsof \
|
||||
locales \
|
||||
# Build essentials (needed for framework stage)
|
||||
build-essential \
|
||||
cmake \
|
||||
perl \
|
||||
patchelf \
|
||||
ccache \
|
||||
git-lfs \
|
||||
# MPI and NUMA
|
||||
libopenmpi-dev \
|
||||
libnuma1 \
|
||||
libnuma-dev \
|
||||
numactl \
|
||||
# transformers multimodal VLM
|
||||
ffmpeg \
|
||||
# InfiniBand/RDMA
|
||||
libibverbs-dev \
|
||||
libibverbs1 \
|
||||
libibumad3 \
|
||||
librdmacm1 \
|
||||
libnl-3-200 \
|
||||
libnl-route-3-200 \
|
||||
libnl-route-3-dev \
|
||||
libnl-3-dev \
|
||||
ibverbs-providers \
|
||||
infiniband-diags \
|
||||
perftest \
|
||||
# Development libraries
|
||||
libgoogle-glog-dev \
|
||||
libgtest-dev \
|
||||
libjsoncpp-dev \
|
||||
libunwind-dev \
|
||||
libboost-all-dev \
|
||||
libssl-dev \
|
||||
libgrpc-dev \
|
||||
libgrpc++-dev \
|
||||
libprotobuf-dev \
|
||||
protobuf-compiler \
|
||||
protobuf-compiler-grpc \
|
||||
pybind11-dev \
|
||||
libhiredis-dev \
|
||||
libcurl4-openssl-dev \
|
||||
libczmq4 \
|
||||
libczmq-dev \
|
||||
libfabric-dev \
|
||||
linux-libc-dev \
|
||||
# Package building tools
|
||||
devscripts \
|
||||
debhelper \
|
||||
fakeroot \
|
||||
dkms \
|
||||
check \
|
||||
libsubunit0 \
|
||||
libsubunit-dev \
|
||||
&& ln -sf /usr/bin/python3.12 /usr/bin/python \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& apt-get clean
|
||||
|
||||
# Replace pip global cache if specified
|
||||
RUN if [ -n "${PIP_DEFAULT_INDEX}" ]; then \
|
||||
python3 -m pip config set global.index-url ${PIP_DEFAULT_INDEX}; \
|
||||
fi
|
||||
|
||||
# GDRCopy installation
|
||||
RUN mkdir -p /tmp/gdrcopy && cd /tmp \
|
||||
&& curl --retry 3 --retry-delay 2 -fsSL -o v${GDRCOPY_VERSION}.tar.gz \
|
||||
https://${GITHUB_ARTIFACTORY}/NVIDIA/gdrcopy/archive/refs/tags/v${GDRCOPY_VERSION}.tar.gz \
|
||||
&& tar -xzf v${GDRCOPY_VERSION}.tar.gz && rm v${GDRCOPY_VERSION}.tar.gz \
|
||||
&& cd gdrcopy-${GDRCOPY_VERSION}/packages \
|
||||
&& CUDA=/usr/local/cuda ./build-deb-packages.sh \
|
||||
&& dpkg -i gdrdrv-dkms_*.deb libgdrapi_*.deb gdrcopy-tests_*.deb gdrcopy_*.deb \
|
||||
&& cd / && rm -rf /tmp/gdrcopy
|
||||
|
||||
# Fix DeepEP IBGDA symlink
|
||||
RUN ln -sf /usr/lib/$(uname -m)-linux-gnu/libmlx5.so.1 /usr/lib/$(uname -m)-linux-gnu/libmlx5.so
|
||||
|
||||
# Set up locale
|
||||
RUN locale-gen en_US.UTF-8
|
||||
ENV LANG=en_US.UTF-8 \
|
||||
LANGUAGE=en_US:en \
|
||||
LC_ALL=en_US.UTF-8
|
||||
|
||||
########################################################
|
||||
########## PARALLEL BUILDER STAGES ####################
|
||||
########################################################
|
||||
#
|
||||
# These stages run IN PARALLEL via BuildKit:
|
||||
#
|
||||
# base
|
||||
# |
|
||||
# +-- torch_deps ------> deepep_builder (needs torch)
|
||||
# | \-> flashinfer_cache (needs flashinfer)
|
||||
# |
|
||||
# +-- devtools_builder (independent)
|
||||
# +-- gateway_builder (independent, only needs gateway source)
|
||||
# |
|
||||
# v
|
||||
# framework (combines all artifacts)
|
||||
#
|
||||
|
||||
########################################################
|
||||
# PARALLEL STAGE 1: Torch/Deps Builder (starts from base)
|
||||
########################################################
|
||||
FROM base AS torch_deps
|
||||
|
||||
ARG CUDA_VERSION
|
||||
ARG BUILD_TYPE
|
||||
ARG SGL_KERNEL_VERSION
|
||||
ARG GITHUB_ARTIFACTORY
|
||||
|
||||
WORKDIR /sgl-workspace
|
||||
|
||||
# Rust toolchain for setuptools-rust extensions (e.g. sglang-grpc).
|
||||
# Requires >= 1.85 (edition 2024). Inherited by framework via FROM torch_deps.
|
||||
ENV PATH="/root/.cargo/bin:${PATH}"
|
||||
RUN curl --proto '=https' --tlsv1.2 --retry 3 --retry-delay 2 -sSf https://sh.rustup.rs \
|
||||
| sh -s -- -y --no-modify-path --profile minimal \
|
||||
&& rustc --version && cargo --version
|
||||
|
||||
# Install sgl-kernel (from pre-built wheel)
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3 -m pip install --upgrade pip setuptools wheel html5lib six \
|
||||
&& case "$CUDA_VERSION" in \
|
||||
12.6.1) CUINDEX=126 ;; \
|
||||
12.8.1) CUINDEX=128 ;; \
|
||||
12.9.1) CUINDEX=129 ;; \
|
||||
13.0.1) CUINDEX=130 ;; \
|
||||
*) echo "Unsupported CUDA version: $CUDA_VERSION" && exit 1 ;; \
|
||||
esac \
|
||||
&& if [ "$CUDA_VERSION" = "12.6.1" ]; then \
|
||||
python3 -m pip install https://${GITHUB_ARTIFACTORY}/sgl-project/whl/releases/download/v${SGL_KERNEL_VERSION}/sglang_kernel-${SGL_KERNEL_VERSION}+cu124-cp310-abi3-manylinux2014_$(uname -m).whl --force-reinstall --no-deps \
|
||||
; \
|
||||
elif [ "$CUDA_VERSION" = "12.8.1" ] || [ "$CUDA_VERSION" = "12.9.1" ]; then \
|
||||
python3 -m pip install https://github.com/sgl-project/whl/releases/download/v${SGL_KERNEL_VERSION}/sglang_kernel-${SGL_KERNEL_VERSION}+cu129-cp310-abi3-manylinux2014_$(uname -m).whl --force-reinstall --no-deps \
|
||||
; \
|
||||
elif [ "$CUDA_VERSION" = "13.0.1" ]; then \
|
||||
# --no-deps prevents pip from pulling torch from default PyPI
|
||||
python3 -m pip install sglang-kernel==${SGL_KERNEL_VERSION} --force-reinstall --no-deps \
|
||||
; \
|
||||
else \
|
||||
echo "Unsupported CUDA version: $CUDA_VERSION" && exit 1 \
|
||||
; \
|
||||
fi
|
||||
|
||||
# Copy dep spec + Rust crate source + proto files. setuptools-rust compiles the
|
||||
# Rust extension during the stub wheel build; the crate's build.rs references
|
||||
# ../../proto for tonic_build. Split from the pip install so source changes to
|
||||
# these paths invalidate the dep-install layer, but Python source changes don't.
|
||||
COPY python/pyproject.toml /tmp/sglang_deps/python/pyproject.toml
|
||||
COPY rust/sglang-grpc /tmp/sglang_deps/rust/sglang-grpc
|
||||
COPY proto /tmp/sglang_deps/proto
|
||||
|
||||
# Install sglang dependencies (torch, transformers, etc.)
|
||||
# Generate constraints.txt to prevent reinstalling these deps in later stages
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
--mount=type=cache,target=/root/.cargo/registry \
|
||||
case "$CUDA_VERSION" in \
|
||||
12.6.1) CUINDEX=126 ;; \
|
||||
12.8.1) CUINDEX=128 ;; \
|
||||
12.9.1) CUINDEX=129 ;; \
|
||||
13.0.1) CUINDEX=130 ;; \
|
||||
*) echo "Unsupported CUDA version: $CUDA_VERSION" && exit 1 ;; \
|
||||
esac \
|
||||
&& cd /tmp/sglang_deps/python \
|
||||
&& mkdir -p sglang \
|
||||
&& touch sglang/__init__.py \
|
||||
&& echo '__version__ = "0.0.0"' > sglang/version.py \
|
||||
&& touch README.md \
|
||||
&& touch LICENSE \
|
||||
&& if [ "${CUDA_VERSION%%.*}" = "12" ]; then \
|
||||
sed -i 's/cuda-python>=13\.0/cuda-python>=12,<13/' pyproject.toml && \
|
||||
sed -i 's/flashinfer_python\[cu13\]/flashinfer_python[cu12]/' pyproject.toml && \
|
||||
sed -i 's/nvidia-cutlass-dsl\[cu13\]/nvidia-cutlass-dsl/' pyproject.toml; \
|
||||
fi \
|
||||
&& python3 -m pip install --extra-index-url https://download.pytorch.org/whl/cu${CUINDEX} ".[${BUILD_TYPE}]" \
|
||||
&& if [ "${CUDA_VERSION%%.*}" = "12" ]; then \
|
||||
pip list --format=freeze | awk -F'==' '/-cu13(==|$)/ {print $1}' \
|
||||
| xargs -r python3 -m pip uninstall -y && \
|
||||
python3 -m pip install --index-url https://download.pytorch.org/whl/cu${CUINDEX} \
|
||||
torch==2.11.0 torchvision==0.26.0 torchaudio==2.11.0 --force-reinstall; \
|
||||
python3 -m pip install https://github.com/sgl-project/whl/releases/download/v${SGL_DEEP_GEMM_VERSION}/sgl_deep_gemm-${SGL_DEEP_GEMM_VERSION}+cu129-py3-none-manylinux2014_$(uname -m).whl --force-reinstall; \
|
||||
fi \
|
||||
&& cd /sgl-workspace \
|
||||
&& rm -rf /tmp/sglang_deps \
|
||||
&& pip freeze | grep -v "^sglang==" > /sgl-workspace/constraints.txt
|
||||
|
||||
########################################################
|
||||
# PARALLEL STAGE 2: DeepEP Builder (needs torch_deps)
|
||||
########################################################
|
||||
FROM torch_deps AS deepep_builder
|
||||
|
||||
ARG CUDA_VERSION
|
||||
ARG BUILD_AND_DOWNLOAD_PARALLEL
|
||||
ARG GRACE_BLACKWELL
|
||||
ARG HOPPER_SBO
|
||||
ARG HOPPER_SBO_DEEPEP_COMMIT
|
||||
ARG DEEPEP_COMMIT
|
||||
ARG GITHUB_ARTIFACTORY
|
||||
|
||||
WORKDIR /build
|
||||
|
||||
# Clone DeepEP
|
||||
RUN set -eux; \
|
||||
if [ "$GRACE_BLACKWELL" = "1" ]; then \
|
||||
if [ "${CUDA_VERSION%%.*}" = "12" ]; then \
|
||||
git clone https://github.com/fzyzcjy/DeepEP.git && \
|
||||
cd DeepEP && \
|
||||
git checkout gb200_blog_part_2 && \
|
||||
sed -i 's/#define NUM_CPU_TIMEOUT_SECS 100/#define NUM_CPU_TIMEOUT_SECS 1000/' csrc/kernels/configs.cuh && \
|
||||
sed -i 's/#define NUM_TIMEOUT_CYCLES 200000000000ull/#define NUM_TIMEOUT_CYCLES 2000000000000ull/' csrc/kernels/configs.cuh && \
|
||||
cd .. ; \
|
||||
else \
|
||||
git clone https://github.com/deepseek-ai/DeepEP.git -b hybrid-ep && \
|
||||
cd DeepEP && \
|
||||
git checkout d28bd676c2120573c9f1425f0c16c39faa4117e6 && \
|
||||
sed -i 's/#define NUM_CPU_TIMEOUT_SECS 100/#define NUM_CPU_TIMEOUT_SECS 1000/' csrc/kernels/configs.cuh && \
|
||||
sed -i 's/#define NUM_TIMEOUT_CYCLES 200000000000ull/#define NUM_TIMEOUT_CYCLES 2000000000000ull/' csrc/kernels/configs.cuh && \
|
||||
cd .. ; \
|
||||
fi; \
|
||||
elif [ "$HOPPER_SBO" = "1" ]; then \
|
||||
git clone https://github.com/deepseek-ai/DeepEP.git -b antgroup-opt && \
|
||||
cd DeepEP && \
|
||||
git checkout ${HOPPER_SBO_DEEPEP_COMMIT} && \
|
||||
sed -i 's/#define NUM_CPU_TIMEOUT_SECS 100/#define NUM_CPU_TIMEOUT_SECS 1000/' csrc/kernels/configs.cuh && \
|
||||
sed -i 's/#define NUM_TIMEOUT_CYCLES 200000000000ull/#define NUM_TIMEOUT_CYCLES 2000000000000ull/' csrc/kernels/configs.cuh && \
|
||||
cd .. ; \
|
||||
else \
|
||||
curl --retry 3 --retry-delay 2 -fsSL -o ${DEEPEP_COMMIT}.zip \
|
||||
https://${GITHUB_ARTIFACTORY}/deepseek-ai/DeepEP/archive/${DEEPEP_COMMIT}.zip && \
|
||||
unzip -q ${DEEPEP_COMMIT}.zip && rm ${DEEPEP_COMMIT}.zip && mv DeepEP-${DEEPEP_COMMIT} DeepEP && cd DeepEP && \
|
||||
sed -i 's/#define NUM_CPU_TIMEOUT_SECS 100/#define NUM_CPU_TIMEOUT_SECS 1000/' csrc/kernels/configs.cuh && \
|
||||
sed -i 's/#define NUM_TIMEOUT_CYCLES 200000000000ull/#define NUM_TIMEOUT_CYCLES 2000000000000ull/' csrc/kernels/configs.cuh && \
|
||||
cd .. ; \
|
||||
fi
|
||||
|
||||
# Build DeepEP wheel
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
cd /build/DeepEP && \
|
||||
case "$CUDA_VERSION" in \
|
||||
12.6.1) \
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='9.0' \
|
||||
;; \
|
||||
12.8.1) \
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='9.0;10.0' \
|
||||
;; \
|
||||
12.9.1|13.0.1) \
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='9.0;10.0;10.3' \
|
||||
;; \
|
||||
*) \
|
||||
echo "Unsupported CUDA version: $CUDA_VERSION" && exit 1 \
|
||||
;; \
|
||||
esac && \
|
||||
if [ "${CUDA_VERSION%%.*}" = "13" ]; then \
|
||||
sed -i "/^ include_dirs = \['csrc\/'\]/a\ include_dirs.append('${CUDA_HOME}/include/cccl')" setup.py; \
|
||||
fi && \
|
||||
TORCH_CUDA_ARCH_LIST="${CHOSEN_TORCH_CUDA_ARCH_LIST}" MAX_JOBS=${BUILD_AND_DOWNLOAD_PARALLEL} \
|
||||
python3 setup.py bdist_wheel -d /wheels
|
||||
|
||||
########################################################
|
||||
# PARALLEL STAGE 3: FlashInfer Cache (needs torch_deps)
|
||||
########################################################
|
||||
FROM torch_deps AS flashinfer_cache
|
||||
|
||||
ARG CUDA_VERSION
|
||||
ARG INSTALL_FLASHINFER_JIT_CACHE
|
||||
ARG FLASHINFER_VERSION
|
||||
|
||||
# Stage jit-cache/cubin artifacts into /flashinfer_jit_output for clean COPY later
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
case "$CUDA_VERSION" in \
|
||||
12.6.1) CUINDEX=126 ;; \
|
||||
12.8.1) CUINDEX=128 ;; \
|
||||
12.9.1) CUINDEX=129 ;; \
|
||||
13.0.1) CUINDEX=130 ;; \
|
||||
*) echo "Unsupported CUDA version: $CUDA_VERSION" && exit 1 ;; \
|
||||
esac \
|
||||
&& mkdir -p /flashinfer_jit_output \
|
||||
# flashinfer-cubin is CUDA-version-agnostic, unlike jit-cache, so its index-url has no cu${CUINDEX} suffix
|
||||
&& python3 -m pip install flashinfer-cubin==${FLASHINFER_VERSION} --index-url https://flashinfer.ai/whl \
|
||||
&& cp -r /usr/local/lib/python3.12/dist-packages/flashinfer_cubin /flashinfer_jit_output/ \
|
||||
&& cp -r /usr/local/lib/python3.12/dist-packages/flashinfer_cubin-*.dist-info /flashinfer_jit_output/ \
|
||||
&& if [ "$INSTALL_FLASHINFER_JIT_CACHE" = "1" ]; then \
|
||||
python3 -m pip install flashinfer-jit-cache==${FLASHINFER_VERSION} --index-url https://flashinfer.ai/whl/cu${CUINDEX} \
|
||||
&& cp -r /usr/local/lib/python3.12/dist-packages/flashinfer_jit_cache /flashinfer_jit_output/ \
|
||||
&& cp -r /usr/local/lib/python3.12/dist-packages/flashinfer_jit_cache-*.dist-info /flashinfer_jit_output/ ; \
|
||||
fi
|
||||
|
||||
########################################################
|
||||
# PARALLEL STAGE 4: Dev Tools Builder (starts from base)
|
||||
########################################################
|
||||
FROM base AS devtools_builder
|
||||
|
||||
ARG GITHUB_ARTIFACTORY
|
||||
|
||||
WORKDIR /tools
|
||||
|
||||
# Minimal apt deps needed for oh-my-zsh install in this stage
|
||||
# Full dev apt packages (gdb, vim, tmux, nsight, etc.) are installed in the framework stage
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=devtools-apt \
|
||||
apt-get update && apt-get install -y --no-install-recommends zsh git \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Download CLI tools (each in its own layer for parallel downloads)
|
||||
RUN curl --retry 3 --retry-delay 2 -LSso /tools/diff-so-fancy \
|
||||
https://${GITHUB_ARTIFACTORY}/so-fancy/diff-so-fancy/releases/download/v1.4.4/diff-so-fancy \
|
||||
&& chmod +x /tools/diff-so-fancy
|
||||
|
||||
RUN curl --retry 3 --retry-delay 2 -LSso /tools/clang-format \
|
||||
https://${GITHUB_ARTIFACTORY}/muttleyxd/clang-tools-static-binaries/releases/download/master-32d3ac78/clang-format-16_linux-amd64 \
|
||||
&& chmod +x /tools/clang-format
|
||||
|
||||
RUN curl --retry 3 --retry-delay 2 -fsSL -o /tmp/clangd.zip \
|
||||
https://${GITHUB_ARTIFACTORY}/clangd/clangd/releases/download/18.1.3/clangd-linux-18.1.3.zip \
|
||||
&& unzip -q /tmp/clangd.zip -d /tmp \
|
||||
&& cp /tmp/clangd_18.1.3/bin/* /tools/ \
|
||||
&& mkdir -p /tools/lib && cp -r /tmp/clangd_18.1.3/lib/* /tools/lib/ \
|
||||
&& rm -rf /tmp/clangd.zip /tmp/clangd_18.1.3
|
||||
|
||||
RUN CMAKE_VERSION=3.31.1 \
|
||||
&& ARCH=$(uname -m) \
|
||||
&& CMAKE_INSTALLER="cmake-${CMAKE_VERSION}-linux-${ARCH}" \
|
||||
&& curl --retry 3 --retry-delay 2 -fsSL -o "/tmp/${CMAKE_INSTALLER}.tar.gz" \
|
||||
"https://${GITHUB_ARTIFACTORY}/Kitware/CMake/releases/download/v${CMAKE_VERSION}/${CMAKE_INSTALLER}.tar.gz" \
|
||||
&& tar -xzf "/tmp/${CMAKE_INSTALLER}.tar.gz" -C /tmp \
|
||||
&& cp -r "/tmp/${CMAKE_INSTALLER}/bin/"* /tools/ \
|
||||
&& mkdir -p /tools/share && cp -r "/tmp/${CMAKE_INSTALLER}/share/"* /tools/share/ \
|
||||
&& rm -rf "/tmp/${CMAKE_INSTALLER}" "/tmp/${CMAKE_INSTALLER}.tar.gz"
|
||||
|
||||
RUN curl --proto '=https' --tlsv1.2 --retry 3 --retry-delay 2 -sSf https://just.systems/install.sh | \
|
||||
sed "s|https://github.com|https://${GITHUB_ARTIFACTORY}|g" | \
|
||||
bash -s -- --tag 1.42.4 --to /tools
|
||||
|
||||
# Install oh-my-zsh and plugins
|
||||
RUN sh -c "$(curl --retry 3 --retry-delay 2 -fsSL https://raw.githubusercontent.com/ohmyzsh/ohmyzsh/master/tools/install.sh)" "" --unattended \
|
||||
&& git clone --depth 1 https://github.com/zsh-users/zsh-autosuggestions ${ZSH_CUSTOM:-/root/.oh-my-zsh/custom}/plugins/zsh-autosuggestions \
|
||||
&& git clone --depth 1 https://github.com/zsh-users/zsh-syntax-highlighting.git ${ZSH_CUSTOM:-/root/.oh-my-zsh/custom}/plugins/zsh-syntax-highlighting
|
||||
|
||||
########################################################
|
||||
# PARALLEL STAGE 5: Gateway Builder (starts from base)
|
||||
########################################################
|
||||
# Builds sgl-model-gateway in isolation so Python-only changes
|
||||
# don't trigger a full Rust recompilation.
|
||||
FROM base AS gateway_builder
|
||||
|
||||
ARG GITHUB_ARTIFACTORY
|
||||
ARG BRANCH_TYPE
|
||||
ARG SGL_VERSION
|
||||
ARG USE_LATEST_SGLANG
|
||||
|
||||
WORKDIR /build
|
||||
|
||||
# Copy ONLY the gateway source (not the full repo)
|
||||
COPY sgl-model-gateway /build/sgl-model-gateway
|
||||
|
||||
# Install Rust, build gateway binary and Python bindings, then clean up Rust toolchain
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
curl --proto '=https' --tlsv1.2 --retry 3 --retry-delay 2 -sSf https://sh.rustup.rs | sh -s -- -y \
|
||||
&& export PATH="/root/.cargo/bin:${PATH}" \
|
||||
&& python3 -m pip install maturin \
|
||||
&& cd /build/sgl-model-gateway/bindings/python \
|
||||
&& ulimit -n 65536 && maturin build --release --features vendored-openssl --out /build/gateway_wheels \
|
||||
&& cd /build/sgl-model-gateway \
|
||||
&& cargo build --release --bin sgl-model-gateway --features vendored-openssl \
|
||||
&& cp target/release/sgl-model-gateway /build/sgl-model-gateway-bin \
|
||||
&& rm -rf /root/.cargo /root/.rustup /build/sgl-model-gateway/target /build/sgl-model-gateway/bindings/python/target
|
||||
|
||||
########################################################
|
||||
########## Final Framework Image ######################
|
||||
########################################################
|
||||
#
|
||||
# Combines all artifacts from parallel builder stages
|
||||
#
|
||||
FROM torch_deps AS framework
|
||||
|
||||
ARG BRANCH_TYPE
|
||||
ARG BUILD_TYPE
|
||||
ARG CUDA_VERSION
|
||||
ARG BUILD_AND_DOWNLOAD_PARALLEL
|
||||
ARG SGL_VERSION
|
||||
ARG USE_LATEST_SGLANG
|
||||
ARG GITHUB_ARTIFACTORY
|
||||
ARG MOONCAKE_VERSION
|
||||
ARG MOONCAKE_COMPILE_ARG
|
||||
ARG MSCCLPP_VERSION
|
||||
|
||||
WORKDIR /sgl-workspace
|
||||
|
||||
# =============================================================================
|
||||
# Copy artifacts from parallel builders
|
||||
# =============================================================================
|
||||
|
||||
# Copy DeepEP wheel and install
|
||||
COPY --from=deepep_builder /wheels /tmp/wheels/deepep
|
||||
COPY --from=deepep_builder /build/DeepEP /sgl-workspace/DeepEP
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
pip install /tmp/wheels/deepep/*.whl && rm -rf /tmp/wheels/deepep
|
||||
|
||||
# Copy flashinfer cubin (always) and jit-cache (if installed) packages
|
||||
COPY --from=flashinfer_cache /flashinfer_jit_output/ /usr/local/lib/python3.12/dist-packages/
|
||||
|
||||
# Copy dev tools
|
||||
COPY --from=devtools_builder /tools/diff-so-fancy /usr/local/bin/
|
||||
COPY --from=devtools_builder /tools/clang-format /usr/local/bin/
|
||||
COPY --from=devtools_builder /tools/clangd /usr/local/bin/
|
||||
COPY --from=devtools_builder /tools/lib /usr/local/lib/
|
||||
COPY --from=devtools_builder /tools/cmake /usr/local/bin/
|
||||
COPY --from=devtools_builder /tools/ctest /usr/local/bin/
|
||||
COPY --from=devtools_builder /tools/cpack /usr/local/bin/
|
||||
COPY --from=devtools_builder /tools/share/cmake-3.31 /usr/local/share/cmake-3.31
|
||||
COPY --from=devtools_builder /tools/just /usr/local/bin/
|
||||
COPY --from=devtools_builder /root/.oh-my-zsh /root/.oh-my-zsh
|
||||
|
||||
# Install dev apt packages (need to re-run since we're in a different stage)
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=framework-apt \
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
gdb \
|
||||
ninja-build \
|
||||
vim \
|
||||
tmux \
|
||||
htop \
|
||||
zsh \
|
||||
tree \
|
||||
silversearcher-ag \
|
||||
cloc \
|
||||
pkg-config \
|
||||
bear \
|
||||
less \
|
||||
rdma-core \
|
||||
openssh-server \
|
||||
gnuplot \
|
||||
infiniband-diags \
|
||||
perftest \
|
||||
ibverbs-providers \
|
||||
libibumad3 \
|
||||
libibverbs1 \
|
||||
libnl-3-200 \
|
||||
libnl-route-3-200 \
|
||||
librdmacm1 \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& apt-get clean
|
||||
|
||||
# Install NVIDIA development tools
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=framework-apt \
|
||||
apt update -y \
|
||||
&& apt install -y --no-install-recommends gnupg \
|
||||
&& echo "deb http://developer.download.nvidia.com/devtools/repos/ubuntu2004/$(if [ "$(uname -m)" = "aarch64" ]; then echo "arm64"; else echo "amd64"; fi) /" | tee /etc/apt/sources.list.d/nvidia-devtools.list \
|
||||
&& apt-key adv --fetch-keys http://developer.download.nvidia.com/compute/cuda/repos/ubuntu1804/$(if [ "$(uname -m)" = "aarch64" ]; then echo "arm64"; else echo "x86_64"; fi)/7fa2af80.pub \
|
||||
&& apt update -y \
|
||||
&& apt install -y --no-install-recommends nsight-systems-cli \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# =============================================================================
|
||||
# Python packages and tools (before source copy for better caching)
|
||||
# =============================================================================
|
||||
|
||||
# Install Mooncake
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
CUDA_MAJOR="${CUDA_VERSION%%.*}" && \
|
||||
if [ "$CUDA_MAJOR" -ge 13 ]; then \
|
||||
python3 -m pip install mooncake-transfer-engine-cuda13==${MOONCAKE_VERSION}; \
|
||||
else \
|
||||
python3 -m pip install mooncake-transfer-engine==${MOONCAKE_VERSION}; \
|
||||
fi
|
||||
|
||||
# Install MSCCL++ Python dependencies and package (builds extension via CMake through pip)
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
git clone --depth=1 --branch ${MSCCLPP_VERSION} https://${GITHUB_ARTIFACTORY}/microsoft/mscclpp.git /tmp/mscclpp \
|
||||
&& case "${CUDA_VERSION}" in \
|
||||
12.*) \
|
||||
CMAKE_ARGS="-DMSCCLPP_BYPASS_GPU_CHECK=ON -DMSCCLPP_USE_CUDA=ON -DMSCCLPP_GPU_ARCHS=80,90,100,100a,103,103a" \
|
||||
python3 -m pip install "/tmp/mscclpp[cuda12]"; \
|
||||
;; \
|
||||
13.*) \
|
||||
CMAKE_ARGS="-DMSCCLPP_BYPASS_GPU_CHECK=ON -DMSCCLPP_USE_CUDA=ON -DMSCCLPP_GPU_ARCHS=80,90,100,100a,103,103a" \
|
||||
python3 -m pip install "/tmp/mscclpp[cuda13]"; \
|
||||
;; \
|
||||
*) \
|
||||
echo "Unsupported CUDA version for MSCCL++: ${CUDA_VERSION}" && exit 1; \
|
||||
;; \
|
||||
esac \
|
||||
&& rm -rf /tmp/mscclpp
|
||||
|
||||
# Install essential Python packages (use constraints to prevent conflicts)
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3 -m pip install -c /sgl-workspace/constraints.txt \
|
||||
datamodel_code_generator \
|
||||
pre-commit \
|
||||
pytest \
|
||||
black \
|
||||
isort \
|
||||
icdiff \
|
||||
uv \
|
||||
wheel \
|
||||
scikit-build-core \
|
||||
py-spy \
|
||||
cubloaty \
|
||||
google-cloud-storage \
|
||||
pandas \
|
||||
matplotlib \
|
||||
tabulate \
|
||||
termplotlib \
|
||||
"runai-model-streamer[s3,gcs,azure]>=0.15.7"
|
||||
|
||||
# Optional ai-dynamo nightly for Dynamo + SGLang integration testing. Gated by
|
||||
# INSTALL_DYNAMO so it lands only in the dev image (release-docker-dev.yml sets
|
||||
# it), not the version-tagged release/runtime images. Empty DYNAMO_VERSION
|
||||
# resolves the latest (highest-version) nightly, matching `pip install --pre`;
|
||||
# set it to pin an exact version for a reproducible build.
|
||||
ARG INSTALL_DYNAMO=0
|
||||
ARG DYNAMO_VERSION=
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
set -eu; \
|
||||
if [ "$INSTALL_DYNAMO" = "1" ]; then \
|
||||
if [ -z "$DYNAMO_VERSION" ]; then \
|
||||
index="$(curl -fsSL --retry 3 --retry-delay 2 https://pypi.nvidia.com/ai-dynamo/)" \
|
||||
|| { echo "ERROR: failed to fetch the ai-dynamo index from pypi.nvidia.com"; exit 1; }; \
|
||||
DYNAMO_VERSION="$(printf '%s\n' "$index" \
|
||||
| grep -oE '[0-9]+\.[0-9]+\.[0-9]+\.dev[0-9]{8}' | sort -Vu | tail -1)"; \
|
||||
case "$DYNAMO_VERSION" in \
|
||||
*.*.*.dev????????) : ;; \
|
||||
*) echo "ERROR: no X.Y.Z.devYYYYMMDD nightly found in the index (format may have changed)"; exit 1 ;; \
|
||||
esac; \
|
||||
fi; \
|
||||
echo "Installing ai-dynamo==${DYNAMO_VERSION}"; \
|
||||
python3 -m pip install --extra-index-url https://pypi.nvidia.com "ai-dynamo==${DYNAMO_VERSION}"; \
|
||||
fi
|
||||
|
||||
# Per-CUDA-major package installs. The `nixl` stub package is needed (it owns
|
||||
# the `nixl` import path) but unconditionally requires nixl-cu12, so we install
|
||||
# it with --no-deps and pair it with the matching nixl-cu12 / nixl-cu13 binary
|
||||
# to avoid shipping wrong-CUDA libs on cu13 images.
|
||||
RUN --mount=type=cache,target=/root/.cache/pip if [ "${CUDA_VERSION%%.*}" = "12" ]; then \
|
||||
python3 -m pip install nixl nixl-cu12 --no-deps ; \
|
||||
python3 -m pip install cuda-python==12.9 ; \
|
||||
elif [ "${CUDA_VERSION%%.*}" = "13" ]; then \
|
||||
python3 -m pip install nixl nixl-cu13 --no-deps ; \
|
||||
python3 -m pip install cuda-python==13.2.0 ; \
|
||||
fi
|
||||
|
||||
# Add yank script
|
||||
COPY --chown=root:root --chmod=755 docker/configs/yank /usr/local/bin/yank
|
||||
|
||||
# These configs are optional; users can override them by mounting their own files
|
||||
COPY docker/configs/opt/.vimrc /opt/sglang/.vimrc
|
||||
COPY docker/configs/opt/.tmux.conf /opt/sglang/.tmux.conf
|
||||
COPY docker/configs/opt/.gitconfig /opt/sglang/.gitconfig
|
||||
|
||||
# Configure development environment
|
||||
COPY docker/configs/.zshrc /root/.zshrc
|
||||
|
||||
# Fix Trivy-reported CVEs
|
||||
# pip: urllib3 (CVE-2025-43859), pillow (CVE-2026-25990)
|
||||
# binutils family: CVE-2025-{1147,1148,3198,5244,5245,7545,7546,8225,11082,11083,11412,11413,11414,11494,11839,11840}
|
||||
# libgnutls30t64: CVE-2025-{9820,14831}
|
||||
# libpam: CVE-2024-10963
|
||||
# libsqlite3-0: CVE-2025-{6965,7709}
|
||||
# libtasn1-6: CVE-2025-13151
|
||||
# dpkg: CVE-2025-6297
|
||||
RUN python3 -m pip install --upgrade "urllib3>=2.6.3" "pillow>=12.1.1"
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=framework-apt \
|
||||
apt-get update && apt-get install -y --only-upgrade \
|
||||
binutils binutils-common binutils-x86-64-linux-gnu libbinutils \
|
||||
libctf0 libctf-nobfd0 libgprofng0 libsframe1 \
|
||||
libgnutls30t64 \
|
||||
libpam-modules libpam-modules-bin libpam-runtime libpam0g \
|
||||
libsqlite3-0 libtasn1-6 \
|
||||
dpkg dpkg-dev libdpkg-perl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# =============================================================================
|
||||
# Copy sglang source and do editable install (LAST for better caching)
|
||||
# =============================================================================
|
||||
|
||||
# Copy local source if building from local
|
||||
FROM scratch AS local_src
|
||||
COPY . /src
|
||||
|
||||
FROM framework AS framework_final
|
||||
|
||||
ARG BRANCH_TYPE
|
||||
ARG BUILD_TYPE
|
||||
ARG CUDA_VERSION
|
||||
ARG SGL_VERSION
|
||||
ARG USE_LATEST_SGLANG
|
||||
ARG INSTALL_DYNAMO=0
|
||||
|
||||
WORKDIR /sgl-workspace
|
||||
|
||||
COPY --from=local_src /src /tmp/local_src
|
||||
RUN if [ "$BRANCH_TYPE" = "local" ]; then \
|
||||
cp -r /tmp/local_src /sgl-workspace/sglang; \
|
||||
elif [ "$USE_LATEST_SGLANG" = "1" ]; then \
|
||||
git clone --depth=1 https://github.com/sgl-project/sglang.git /sgl-workspace/sglang; \
|
||||
elif [ -z "$SGL_VERSION" ]; then \
|
||||
echo "ERROR: SGL_VERSION must be set when USE_LATEST_SGLANG=0 and BRANCH_TYPE!=local" && exit 1; \
|
||||
else \
|
||||
git clone --depth=1 --branch v${SGL_VERSION} https://github.com/sgl-project/sglang.git /sgl-workspace/sglang; \
|
||||
fi \
|
||||
&& rm -rf /tmp/local_src
|
||||
|
||||
# Editable install (fast - dependencies already installed via constraints)
|
||||
# Clean up __pycache__/tests/pyc in same RUN to avoid writing ~28k files to layer
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
cd /sgl-workspace/sglang \
|
||||
&& if [ "${CUDA_VERSION%%.*}" = "12" ]; then \
|
||||
sed -i 's/cuda-python>=13\.0/cuda-python>=12,<13/' python/pyproject.toml && \
|
||||
sed -i 's/flashinfer_python\[cu13\]/flashinfer_python[cu12]/' python/pyproject.toml && \
|
||||
sed -i 's/nvidia-cutlass-dsl\[cu13\]/nvidia-cutlass-dsl/' python/pyproject.toml; \
|
||||
fi \
|
||||
&& python3 -m pip install --no-deps -e "python[${BUILD_TYPE}]" \
|
||||
&& kernels lock python \
|
||||
&& ( success=0; \
|
||||
# aarch64: kernels-community/sgl-flash-attn3 ships no arm variants; JIT-compile at runtime.
|
||||
# Remove this branch once arm cubins are published upstream.
|
||||
if [ "$(uname -m)" = "aarch64" ]; then \
|
||||
echo "Skipping kernels-community/sgl-flash-attn3 cubin download on aarch64 (no variants published upstream); kernels will be JIT-compiled at runtime"; \
|
||||
success=1; \
|
||||
else \
|
||||
for i in 1 2 3; do \
|
||||
echo "Attempt $i/3: downloading sgl-kernel cubins..." && \
|
||||
kernels download python && \
|
||||
success=1 && break; \
|
||||
echo "sgl-kernel cubin download failed, retrying in 30s..." && sleep 30; \
|
||||
done; \
|
||||
# x86: if no prebuilt sgl-flash-attn3 variant matches this torch+CUDA \
|
||||
# combo (e.g. cu129 publishes no torch>=2.10 cubin), fall back to \
|
||||
# runtime JIT instead of failing the build, mirroring the aarch64 branch. \
|
||||
if [ "$success" != "1" ]; then \
|
||||
echo "WARNING: no matching sgl-flash-attn3 cubin variant for this torch+CUDA; kernels will be JIT-compiled at runtime"; \
|
||||
success=1; \
|
||||
fi; \
|
||||
fi; \
|
||||
[ "$success" = "1" ] ) \
|
||||
&& mkdir -p /root/.cache/huggingface /root/.cache/sglang \
|
||||
&& ( if [ -f python/kernels.lock ]; then mv python/kernels.lock /root/.cache/sglang/; fi ) \
|
||||
&& ( find /usr/local/lib/python3.12/dist-packages -type d -name "__pycache__" -exec rm -rf {} + 2>/dev/null || true )
|
||||
|
||||
|
||||
# Install pre-built gateway artifacts from parallel builder
|
||||
COPY --from=gateway_builder /build/sgl-model-gateway-bin /usr/local/bin/sgl-model-gateway
|
||||
COPY --from=gateway_builder /build/gateway_wheels /tmp/gateway_wheels
|
||||
# When ai-dynamo is installed (dev image), reinstall the gateway wheel with
|
||||
# --no-deps so it cannot re-resolve and replace Dynamo's shared dependencies.
|
||||
# Otherwise resolve the gateway's own dependencies normally.
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
if [ "$INSTALL_DYNAMO" = "1" ]; then \
|
||||
python3 -m pip install --force-reinstall --no-deps /tmp/gateway_wheels/*.whl; \
|
||||
else \
|
||||
python3 -m pip install --force-reinstall /tmp/gateway_wheels/*.whl; \
|
||||
fi \
|
||||
&& rm -rf /tmp/gateway_wheels
|
||||
|
||||
# Set workspace directory
|
||||
WORKDIR /sgl-workspace/sglang
|
||||
|
||||
# Keep build provenance at the end so metadata changes do not invalidate build layers.
|
||||
ARG SGLANG_BUILD_COMMIT=unknown
|
||||
ARG SGLANG_BUILD_URL=
|
||||
ARG SGLANG_IMAGE_TAG=local/sglang:dev
|
||||
ENV SGLANG_BUILD_COMMIT=${SGLANG_BUILD_COMMIT:-unknown} \
|
||||
SGLANG_BUILD_URL=${SGLANG_BUILD_URL:-} \
|
||||
SGLANG_IMAGE_TAG=${SGLANG_IMAGE_TAG:-local/sglang:dev}
|
||||
LABEL org.opencontainers.image.source="https://github.com/sgl-project/sglang" \
|
||||
org.opencontainers.image.revision="${SGLANG_BUILD_COMMIT}" \
|
||||
org.opencontainers.image.version="${SGLANG_IMAGE_TAG}" \
|
||||
org.opencontainers.image.url="${SGLANG_BUILD_URL}" \
|
||||
ai.sglang.build.commit="${SGLANG_BUILD_COMMIT}" \
|
||||
ai.sglang.build.url="${SGLANG_BUILD_URL}" \
|
||||
ai.sglang.image.tag="${SGLANG_IMAGE_TAG}"
|
||||
|
||||
########################################################
|
||||
########## Runtime Image ##############################
|
||||
########################################################
|
||||
#
|
||||
# PURPOSE: Production runtime environment with JIT support
|
||||
#
|
||||
# This stage creates a production-ready image containing:
|
||||
# - Pre-compiled SGLang and DeepEP components
|
||||
# - Full CUDA toolchain for JIT compilation (DeepGEMM, Triton, FlashInfer)
|
||||
# - Optimized for inference workloads and deployment
|
||||
# - Smaller than framework (no dev tools like vim, tmux, nsight, etc.)
|
||||
#
|
||||
# Use this stage when you need:
|
||||
# - Production deployment of SGLang
|
||||
# - JIT compilation support for FP8/microscaling kernels
|
||||
# - Ready-to-run inference server environment
|
||||
#
|
||||
# Note: Uses devel base for complete NVCC toolchain required by DeepGEMM JIT
|
||||
FROM nvidia/cuda:${CUDA_VERSION}-cudnn-devel-ubuntu24.04 AS runtime
|
||||
|
||||
ARG CUDA_VERSION
|
||||
ARG TARGETARCH
|
||||
ARG GDRCOPY_VERSION=2.5.1
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive \
|
||||
CUDA_HOME=/usr/local/cuda \
|
||||
GDRCOPY_HOME=/usr/src/gdrdrv-${GDRCOPY_VERSION}/
|
||||
|
||||
# Add GKE default lib and bin locations + CUDA compiler paths for FlashInfer JIT
|
||||
ENV PATH="${PATH}:/usr/local/nvidia/bin:/usr/local/cuda/bin:/usr/local/cuda/nvvm/bin" \
|
||||
LD_LIBRARY_PATH="${LD_LIBRARY_PATH}:/usr/local/nvidia/lib:/usr/local/nvidia/lib64"
|
||||
|
||||
# Install runtime dependencies (devel base provides gcc/g++/build tools)
|
||||
# Python 3.12 ships in Ubuntu 24.04 main, so no deadsnakes PPA needed.
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=runtime-apt \
|
||||
apt-get update && apt-get install -y --no-install-recommends --allow-change-held-packages \
|
||||
# Python runtime
|
||||
python3.12-full \
|
||||
python3.12-dev \
|
||||
wget \
|
||||
# Core system utilities
|
||||
ca-certificates \
|
||||
netcat-openbsd \
|
||||
curl \
|
||||
git \
|
||||
# Runtime libraries
|
||||
libopenmpi3 \
|
||||
libnuma1 \
|
||||
libibverbs1 \
|
||||
libibumad3 \
|
||||
librdmacm1 \
|
||||
libnl-3-200 \
|
||||
libnl-route-3-200 \
|
||||
ibverbs-providers \
|
||||
libgoogle-glog0v6t64 \
|
||||
libunwind8 \
|
||||
libboost-system1.83.0 \
|
||||
libboost-thread1.83.0 \
|
||||
libboost-filesystem1.83.0 \
|
||||
libgrpc++1.51t64 \
|
||||
libprotobuf32t64 \
|
||||
libhiredis1.1.0 \
|
||||
libcurl4 \
|
||||
libczmq4 \
|
||||
libfabric1 \
|
||||
libssl3 \
|
||||
# RDMA runtime
|
||||
rdma-core \
|
||||
infiniband-diags \
|
||||
perftest \
|
||||
# Build tools for JIT compilation
|
||||
ninja-build \
|
||||
# NCCL packages needed for pynccl_allocator JIT compilation (-lnccl)
|
||||
libnccl2 \
|
||||
libnccl-dev \
|
||||
# GPG key verification
|
||||
gnupg2 \
|
||||
linux-libc-dev \
|
||||
&& update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.12 2 \
|
||||
&& update-alternatives --set python3 /usr/bin/python3.12 \
|
||||
&& ln -sf /usr/bin/python3.12 /usr/bin/python \
|
||||
&& wget -q https://bootstrap.pypa.io/get-pip.py \
|
||||
&& python3 get-pip.py --break-system-packages \
|
||||
&& rm get-pip.py \
|
||||
# Allow pip to install packages globally (PEP 668 workaround for Ubuntu 24.04)
|
||||
&& python3 -m pip config set global.break-system-packages true \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& apt-get clean
|
||||
|
||||
# Set up locale
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends locales \
|
||||
&& locale-gen en_US.UTF-8 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
ENV LANG=en_US.UTF-8 \
|
||||
LANGUAGE=en_US:en \
|
||||
LC_ALL=en_US.UTF-8
|
||||
|
||||
# Fix Trivy-reported CVEs (see framework stage for full CVE list)
|
||||
RUN --mount=type=cache,target=/var/cache/apt,id=runtime-apt \
|
||||
apt-get update && apt-get install -y --only-upgrade \
|
||||
binutils binutils-common binutils-x86-64-linux-gnu libbinutils \
|
||||
libctf0 libctf-nobfd0 libgprofng0 libsframe1 \
|
||||
libgnutls30t64 \
|
||||
libpam-modules libpam-modules-bin libpam-runtime libpam0g \
|
||||
libsqlite3-0 libtasn1-6 \
|
||||
dpkg dpkg-dev libdpkg-perl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Copy Python site-packages from framework (already cleaned of __pycache__/tests/pyc files)
|
||||
COPY --from=framework_final /usr/local/lib/python3.12/dist-packages /usr/local/lib/python3.12/dist-packages
|
||||
|
||||
# Copy SGLang workspace
|
||||
COPY --from=framework_final /sgl-workspace /sgl-workspace
|
||||
|
||||
# Copy sgl-model-gateway binary
|
||||
COPY --from=framework_final /usr/local/bin/sgl-model-gateway /usr/local/bin/sgl-model-gateway
|
||||
|
||||
# Copy sglang binary
|
||||
COPY --from=framework_final /usr/local/bin/sglang /usr/local/bin/sglang
|
||||
|
||||
# Copy py-spy binary
|
||||
COPY --from=framework_final /usr/local/bin/py-spy /usr/local/bin/py-spy
|
||||
|
||||
# Copy cache for kernels from kernels community
|
||||
COPY --from=framework_final /root/.cache/huggingface /root/.cache/huggingface
|
||||
COPY --from=framework_final /root/.cache/sglang /root/.cache/sglang
|
||||
|
||||
# Copy GDRCopy runtime libraries (but not the build artifacts)
|
||||
COPY --from=framework_final /usr/lib/libgdrapi.so* /usr/lib/
|
||||
COPY --from=framework_final /usr/bin/gdrcopy_* /usr/bin/
|
||||
COPY --from=framework_final /usr/src/gdrdrv-2.5.1 /usr/src/gdrdrv-2.5.1
|
||||
|
||||
# Fix DeepEP IBGDA symlink in runtime
|
||||
RUN ln -sf /usr/lib/$(uname -m)-linux-gnu/libmlx5.so.1 /usr/lib/$(uname -m)-linux-gnu/libmlx5.so
|
||||
|
||||
WORKDIR /sgl-workspace/sglang
|
||||
|
||||
# Keep build provenance at the end so metadata changes do not invalidate build layers.
|
||||
ARG SGLANG_BUILD_COMMIT=unknown
|
||||
ARG SGLANG_BUILD_URL=
|
||||
ARG SGLANG_IMAGE_TAG=local/sglang:dev
|
||||
ENV SGLANG_BUILD_COMMIT=${SGLANG_BUILD_COMMIT:-unknown} \
|
||||
SGLANG_BUILD_URL=${SGLANG_BUILD_URL:-} \
|
||||
SGLANG_IMAGE_TAG=${SGLANG_IMAGE_TAG:-local/sglang:dev}
|
||||
LABEL org.opencontainers.image.source="https://github.com/sgl-project/sglang" \
|
||||
org.opencontainers.image.revision="${SGLANG_BUILD_COMMIT}" \
|
||||
org.opencontainers.image.version="${SGLANG_IMAGE_TAG}" \
|
||||
org.opencontainers.image.url="${SGLANG_BUILD_URL}" \
|
||||
ai.sglang.build.commit="${SGLANG_BUILD_COMMIT}" \
|
||||
ai.sglang.build.url="${SGLANG_BUILD_URL}" \
|
||||
ai.sglang.image.tag="${SGLANG_IMAGE_TAG}"
|
||||
|
||||
# Default command
|
||||
CMD ["/bin/bash"]
|
||||
@@ -0,0 +1,52 @@
|
||||
FROM ubuntu:24.04
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
ARG SGLANG_REPO=https://github.com/sgl-project/sglang.git
|
||||
ARG VER_SGLANG=main
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get full-upgrade -y && \
|
||||
DEBIAN_FRONTEND=noninteractive apt-get install --no-install-recommends -y \
|
||||
ca-certificates \
|
||||
git \
|
||||
curl \
|
||||
wget \
|
||||
vim \
|
||||
gcc \
|
||||
g++ \
|
||||
make \
|
||||
cmake \
|
||||
libsqlite3-dev \
|
||||
google-perftools \
|
||||
libtbb-dev \
|
||||
libnuma-dev \
|
||||
numactl
|
||||
|
||||
WORKDIR /opt
|
||||
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh && \
|
||||
source $HOME/.local/bin/env && \
|
||||
uv venv --python 3.12
|
||||
|
||||
RUN echo -e '[[index]]\nname = "torch"\nurl = "https://download.pytorch.org/whl/cpu"\n\n[[index]]\nname = "torchvision"\nurl = "https://download.pytorch.org/whl/cpu"\n\n[[index]]\nname = "torchaudio"\nurl = "https://download.pytorch.org/whl/cpu"\n\n[[index]]\nname = "triton"\nurl = "https://download.pytorch.org/whl/cpu"' > .venv/uv.toml
|
||||
|
||||
ENV UV_CONFIG_FILE=/opt/.venv/uv.toml
|
||||
ENV CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
|
||||
WORKDIR /sgl-workspace
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/.venv/bin/activate && \
|
||||
git clone ${SGLANG_REPO} sglang && \
|
||||
cd sglang && \
|
||||
git checkout ${VER_SGLANG} && \
|
||||
cd python && \
|
||||
cp pyproject_cpu.toml pyproject.toml && \
|
||||
uv pip install . && \
|
||||
cd ../sgl-kernel && \
|
||||
cp pyproject_cpu.toml pyproject.toml && \
|
||||
uv pip install .
|
||||
|
||||
ENV SGLANG_USE_CPU_ENGINE=1
|
||||
RUN echo 'source /opt/.venv/bin/activate' >> /root/.bashrc
|
||||
|
||||
WORKDIR /sgl-workspace/sglang
|
||||
@@ -0,0 +1,35 @@
|
||||
services:
|
||||
sglang:
|
||||
image: lmsysorg/sglang:latest
|
||||
container_name: sglang
|
||||
volumes:
|
||||
- ${HOME}/.cache/huggingface:/root/.cache/huggingface
|
||||
# If you use modelscope, you need mount this directory
|
||||
# - ${HOME}/.cache/modelscope:/root/.cache/modelscope
|
||||
restart: always
|
||||
network_mode: host # required by RDMA
|
||||
privileged: true # required by RDMA
|
||||
# Or you can only publish port 30000
|
||||
# ports:
|
||||
# - 30000:30000
|
||||
environment:
|
||||
- HF_TOKEN=<secret>
|
||||
# if you use modelscope to download model, you need set this environment
|
||||
# - SGLANG_USE_MODELSCOPE=true
|
||||
entrypoint: python3 -m sglang.launch_server
|
||||
command: --model-path meta-llama/Llama-3.1-8B-Instruct
|
||||
--host 0.0.0.0
|
||||
--port 30000
|
||||
ulimits:
|
||||
memlock: -1
|
||||
stack: 67108864
|
||||
ipc: host
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -f http://localhost:30000/health || exit 1"]
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: 1
|
||||
capabilities: [gpu]
|
||||
@@ -0,0 +1,27 @@
|
||||
export ZSH="/root/.oh-my-zsh"
|
||||
|
||||
# Theme
|
||||
ZSH_THEME="robbyrussell"
|
||||
|
||||
# Plugins
|
||||
plugins=(
|
||||
git
|
||||
z
|
||||
zsh-autosuggestions
|
||||
zsh-syntax-highlighting
|
||||
)
|
||||
|
||||
source $ZSH/oh-my-zsh.sh
|
||||
|
||||
# Aliases
|
||||
alias ll='ls -alF'
|
||||
alias la='ls -A'
|
||||
alias l='ls -CF'
|
||||
alias vi='vim'
|
||||
|
||||
# Enhanced history
|
||||
HISTSIZE=10000
|
||||
SAVEHIST=10000
|
||||
setopt HIST_IGNORE_ALL_DUPS
|
||||
setopt HIST_FIND_NO_DUPS
|
||||
setopt INC_APPEND_HISTORY
|
||||
@@ -0,0 +1,30 @@
|
||||
[core]
|
||||
editor = vim
|
||||
whitespace = fix,-indent-with-non-tab,trailing-space,cr-at-eol
|
||||
pager = diff-so-fancy | less --tabs=4 -RFX
|
||||
|
||||
[color]
|
||||
ui = true
|
||||
|
||||
[color "diff-highlight"]
|
||||
oldNormal = red bold
|
||||
oldHighlight = red bold 52
|
||||
newNormal = green bold
|
||||
newHighlight = green bold 22
|
||||
|
||||
[color "diff"]
|
||||
meta = 11
|
||||
frag = magenta bold
|
||||
commit = yellow bold
|
||||
old = red bold
|
||||
new = green bold
|
||||
whitespace = red reverse
|
||||
|
||||
[alias]
|
||||
lg = log --color --graph --pretty=format:'%Cred%h%Creset - %s %Cgreen(%cr) %C(bold blue)<%an>%Creset%C(auto)%d%Creset' --abbrev-commit --
|
||||
|
||||
[http]
|
||||
sslVerify = false
|
||||
|
||||
[pull]
|
||||
rebase = true
|
||||
@@ -0,0 +1,27 @@
|
||||
# Pane border styling
|
||||
set -g pane-border-style fg='#742727',bg=black
|
||||
set -g pane-active-border-style fg=red,bg=black
|
||||
|
||||
# Status bar styling
|
||||
set -g status-style bg='#0C8A92',fg=black
|
||||
|
||||
# Change prefix key to backtick
|
||||
set-option -g prefix `
|
||||
unbind C-b
|
||||
bind-key ` send-prefix
|
||||
|
||||
# Split panes using - and = with current path
|
||||
unbind '"'
|
||||
bind - splitw -v -c '#{pane_current_path}'
|
||||
unbind '%'
|
||||
bind = splitw -h -c '#{pane_current_path}'
|
||||
|
||||
# Vi mode settings
|
||||
bind-key -T copy-mode-vi Y send-keys -X copy-pipe 'yank > #{pane_tty}'
|
||||
set-window-option -g mode-keys vi
|
||||
|
||||
# Other settings
|
||||
set-option -g escape-time 0
|
||||
set-option -g base-index 1
|
||||
set-window-option -g mouse on
|
||||
set -g history-limit 100000
|
||||
@@ -0,0 +1,45 @@
|
||||
function! Yank(text) abort
|
||||
let escape = system('yank', a:text)
|
||||
if v:shell_error
|
||||
echoerr escape
|
||||
else
|
||||
call writefile([escape], '/dev/tty', 'b')
|
||||
endif
|
||||
endfunction
|
||||
|
||||
noremap <silent> <Leader>y y:<C-U>call Yank(@0)<CR>
|
||||
|
||||
" automatically run yank(1) whenever yanking in Vim
|
||||
function! CopyYank() abort
|
||||
call Yank(join(v:event.regcontents, "\n"))
|
||||
endfunction
|
||||
|
||||
autocmd TextYankPost * call CopyYank()
|
||||
|
||||
" Basic settings
|
||||
set number
|
||||
syntax on
|
||||
set mouse=a
|
||||
filetype indent on
|
||||
|
||||
" Indentation
|
||||
set autoindent nosmartindent
|
||||
set smarttab
|
||||
set expandtab
|
||||
set shiftwidth=4
|
||||
set softtabstop=4
|
||||
|
||||
" Visual guides
|
||||
set colorcolumn=120
|
||||
highlight ColorColumn ctermbg=5
|
||||
|
||||
" Status line
|
||||
set laststatus=2
|
||||
set statusline=%<%f\ %h%m%r%=%{\"[\".(&fenc==\"\"?&enc:&fenc).((exists(\"+bomb\")\ &&\ &bomb)?\",B\":\"\").\"]\ \"}%k\ %-14.(%l,%c%V%)\ %P
|
||||
|
||||
" Backspace behavior
|
||||
set backspace=2
|
||||
|
||||
" Encoding
|
||||
set encoding=utf-8
|
||||
set fileencoding=utf-8
|
||||
Executable
+12
@@ -0,0 +1,12 @@
|
||||
#!/bin/bash
|
||||
put() {
|
||||
esc=$1
|
||||
test -n "$TMUX" -o -z "${TERM##screen*}" && esc="\033Ptmux;\033$esc\033\\"
|
||||
printf "$esc"
|
||||
}
|
||||
put "\033]52;c;!\a"
|
||||
buf=$( cat "$@" )
|
||||
len=$( printf %s "$buf" | wc -c ) max=74994
|
||||
test $len -gt $max && echo "$0: input is $(( len - max )) bytes too long" >&2
|
||||
put "\033]52;c;$( printf %s "$buf" | head -c $max | base64 | tr -d '\r\n' )\a"
|
||||
test -n "$TMUX" && tmux set-buffer "$buf" ||:
|
||||
@@ -0,0 +1,77 @@
|
||||
######################## BASE IMAGE ##########################
|
||||
FROM ubuntu:24.04 AS base
|
||||
|
||||
ARG PYTHON_VERSION=3.12
|
||||
|
||||
# set the environment variables
|
||||
ENV PATH="/root/.local/bin:${PATH}"
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# uv environment variables
|
||||
ENV UV_HTTP_TIMEOUT=500
|
||||
ENV VIRTUAL_ENV="/opt/venv"
|
||||
ENV UV_PYTHON_INSTALL_DIR=/opt/uv/python
|
||||
ENV UV_LINK_MODE="copy"
|
||||
ENV PATH="$VIRTUAL_ENV/bin:$PATH"
|
||||
|
||||
|
||||
# install dependencies
|
||||
RUN apt update -y \
|
||||
&& apt install -y curl \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& apt clean
|
||||
|
||||
# install uv
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
|
||||
# install python
|
||||
RUN uv venv --python ${PYTHON_VERSION} --seed ${VIRTUAL_ENV}
|
||||
|
||||
FROM scratch AS local_src
|
||||
COPY . /src
|
||||
|
||||
######################### BUILD IMAGE #########################
|
||||
FROM base AS build-image
|
||||
|
||||
# set the environment variables
|
||||
ENV PATH="/root/.cargo/bin:${PATH}"
|
||||
|
||||
# install dependencies
|
||||
RUN apt update -y \
|
||||
&& apt install -y git build-essential libssl-dev pkg-config protobuf-compiler \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& apt clean
|
||||
|
||||
# install rustup from rustup.rs
|
||||
RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y \
|
||||
&& rustc --version && cargo --version && protoc --version
|
||||
|
||||
# copy source code
|
||||
COPY --from=local_src /src /opt/sglang
|
||||
|
||||
# working directory
|
||||
WORKDIR /opt/sglang/sgl-model-gateway
|
||||
|
||||
# install maturin and build the wheel with vendored OpenSSL
|
||||
RUN uv pip install maturin \
|
||||
&& cargo clean \
|
||||
&& rm -rf bindings/python/dist/ \
|
||||
&& cd bindings/python \
|
||||
&& ulimit -n 65536 && maturin build --release --features vendored-openssl --out dist \
|
||||
&& rm -rf /root/.cache
|
||||
|
||||
######################### ROUTER IMAGE #########################
|
||||
FROM base AS router-image
|
||||
|
||||
# Copy the built package from the build image
|
||||
COPY --from=build-image /opt/sglang/sgl-model-gateway/bindings/python/dist/*.whl dist/
|
||||
|
||||
# Build the package and install
|
||||
RUN uv pip install --force-reinstall dist/*.whl
|
||||
|
||||
# Clean up unnecessary files to reduce the image size
|
||||
RUN rm -rf /root/.cache dist/ \
|
||||
&& apt purge -y --auto-remove curl
|
||||
|
||||
# Set the entrypoint to the main command
|
||||
ENTRYPOINT ["python3", "-m", "sglang_router.launch_router"]
|
||||
@@ -0,0 +1,103 @@
|
||||
# Two Nodes Sglang example
|
||||
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: distributed-sglang
|
||||
spec:
|
||||
replicas: 2 # number of nodes/pods to run distributed sglang
|
||||
selector:
|
||||
matchLabels:
|
||||
app: distributed-sglang
|
||||
serviceName: ""
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: distributed-sglang
|
||||
spec:
|
||||
containers:
|
||||
- name: sglang-container
|
||||
image: docker.io/lmsysorg/sglang:latest
|
||||
imagePullPolicy: Always # image may be replaced by official CI versioned image
|
||||
command:
|
||||
- /bin/bash
|
||||
- -c
|
||||
# please modify the sglang serving arguments below, as necessary.
|
||||
# NOTE: the --expert-parallel-size is for MoE model like DeepSeek-R1
|
||||
args:
|
||||
- |
|
||||
python3 -m sglang.launch_server \
|
||||
--model /llm-folder \
|
||||
--dist-init-addr sglang-master-pod:5000 \
|
||||
--tensor-parallel-size 16 \
|
||||
--nnodes 2 \
|
||||
--node-rank $POD_INDEX \
|
||||
--trust-remote-code \
|
||||
--host 0.0.0.0 \
|
||||
--port 8000 \
|
||||
--enable-metrics \
|
||||
--expert-parallel-size 16
|
||||
env:
|
||||
- name: POD_INDEX # reflects the node-rank
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
apiVersion: v1
|
||||
fieldPath: metadata.labels['apps.kubernetes.io/pod-index']
|
||||
- name: NCCL_DEBUG
|
||||
value: INFO
|
||||
resources:
|
||||
limits:
|
||||
nvidia.com/gpu: "8"
|
||||
requests:
|
||||
volumeMounts:
|
||||
- mountPath: /dev/shm
|
||||
name: dshm
|
||||
- mountPath: /llm-folder
|
||||
name: llm
|
||||
securityContext:
|
||||
privileged: true # to leverage RDMA/InfiniBand device, co-work with HostNetwork=true
|
||||
hostNetwork: true
|
||||
volumes:
|
||||
- emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 10Gi
|
||||
name: dshm
|
||||
- hostPath:
|
||||
path: /llm-folder # replace with PVC or hostPath with your model weights
|
||||
type: DirectoryOrCreate
|
||||
name: llm
|
||||
#- persistentVolumeClaim:
|
||||
# claimName: llm-pvc
|
||||
# name: llm
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: sglang-master-pod
|
||||
spec:
|
||||
type: ClusterIP
|
||||
selector:
|
||||
app: distributed-sglang
|
||||
apps.kubernetes.io/pod-index: "0"
|
||||
ports:
|
||||
- name: dist-port
|
||||
port: 5000
|
||||
targetPort: 5000
|
||||
---
|
||||
# the serving service
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: sglang-serving-on-master
|
||||
spec:
|
||||
type: NodePort
|
||||
selector:
|
||||
app: distributed-sglang
|
||||
apps.kubernetes.io/pod-index: "0"
|
||||
ports:
|
||||
- name: serving
|
||||
port: 8000
|
||||
targetPort: 8000
|
||||
- name: metrics
|
||||
port: 8080
|
||||
targetPort: 8080
|
||||
@@ -0,0 +1,117 @@
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: llama-31-8b-sglang
|
||||
spec:
|
||||
accessModes:
|
||||
- ReadWriteMany
|
||||
resources:
|
||||
requests:
|
||||
storage: 30Gi
|
||||
storageClassName: default # change this to your preferred storage class
|
||||
volumeMode: Filesystem
|
||||
---
|
||||
apiVersion: node.k8s.io/v1
|
||||
kind: RuntimeClass
|
||||
metadata:
|
||||
name: nvidia
|
||||
handler: nvidia
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: meta-llama-31-8b-instruct-sglang
|
||||
spec:
|
||||
replicas: 1
|
||||
strategy:
|
||||
type: Recreate
|
||||
selector:
|
||||
matchLabels:
|
||||
app: meta-llama-31-8b-instruct-sglang
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: meta-llama-31-8b-instruct-sglang
|
||||
model: meta-llama-31-8b-instruct
|
||||
engine: sglang
|
||||
spec:
|
||||
restartPolicy: Always
|
||||
runtimeClassName: nvidia
|
||||
containers:
|
||||
- name: meta-llama-31-8b-instruct-sglang
|
||||
image: docker.io/lmsysorg/sglang:latest
|
||||
imagePullPolicy: Always # IfNotPresent or Never
|
||||
ports:
|
||||
- containerPort: 30000
|
||||
command: ["python3", "-m", "sglang.launch_server"]
|
||||
args:
|
||||
[
|
||||
"--model-path",
|
||||
"meta-llama/Llama-3.1-8B-Instruct",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"30000",
|
||||
]
|
||||
env:
|
||||
- name: HF_TOKEN
|
||||
value: <secret>
|
||||
resources:
|
||||
limits:
|
||||
nvidia.com/gpu: 1
|
||||
cpu: 8
|
||||
memory: 40Gi
|
||||
requests:
|
||||
cpu: 2
|
||||
memory: 16Gi
|
||||
nvidia.com/gpu: 1
|
||||
volumeMounts:
|
||||
- name: shm
|
||||
mountPath: /dev/shm
|
||||
- name: hf-cache
|
||||
mountPath: /root/.cache/huggingface
|
||||
- name: localtime
|
||||
mountPath: /etc/localtime
|
||||
readOnly: true
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
port: 30000
|
||||
initialDelaySeconds: 120
|
||||
periodSeconds: 15
|
||||
timeoutSeconds: 10
|
||||
failureThreshold: 3
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /health_generate
|
||||
port: 30000
|
||||
initialDelaySeconds: 120
|
||||
periodSeconds: 15
|
||||
timeoutSeconds: 10
|
||||
failureThreshold: 3
|
||||
successThreshold: 1
|
||||
volumes:
|
||||
- name: shm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 10Gi
|
||||
- name: hf-cache
|
||||
persistentVolumeClaim:
|
||||
claimName: llama-31-8b-sglang
|
||||
- name: localtime
|
||||
hostPath:
|
||||
path: /etc/localtime
|
||||
type: File
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: meta-llama-31-8b-instruct-sglang
|
||||
spec:
|
||||
selector:
|
||||
app: meta-llama-31-8b-instruct-sglang
|
||||
ports:
|
||||
- protocol: TCP
|
||||
port: 80 # port on host
|
||||
targetPort: 30000 # port in container
|
||||
type: LoadBalancer # change to ClusterIP if needed
|
||||
@@ -0,0 +1,112 @@
|
||||
ARG CANN_VERSION=9.0.0
|
||||
ARG DEVICE_TYPE=a3
|
||||
ARG OS=ubuntu22.04
|
||||
ARG PYTHON_VERSION=py3.11
|
||||
|
||||
FROM quay.io/ascend/cann:$CANN_VERSION-$DEVICE_TYPE-$OS-$PYTHON_VERSION
|
||||
|
||||
# Update pip & apt sources
|
||||
ARG TARGETARCH
|
||||
ARG CANN_VERSION
|
||||
ARG DEVICE_TYPE
|
||||
ARG PIP_INDEX_URL="https://pypi.org/simple/"
|
||||
ARG APTMIRROR=""
|
||||
ARG PYTORCH_VERSION="2.10.0"
|
||||
ARG TORCHVISION_VERSION="0.25.0"
|
||||
ARG TORCHAUDIO_VERSION="2.10.0"
|
||||
ARG PTA_URL_ARM64="https://gitcode.com/Ascend/pytorch/releases/download/v26.0.0-pytorch2.10.0/torch_npu-2.10.0-cp311-cp311-manylinux_2_28_aarch64.whl"
|
||||
ARG PTA_URL_AMD64="https://gitcode.com/Ascend/pytorch/releases/download/v26.0.0-pytorch2.10.0/torch_npu-2.10.0-cp311-cp311-manylinux_2_28_x86_64.whl"
|
||||
ARG SGLANG_TAG=main
|
||||
ARG ASCEND_CANN_PATH=/usr/local/Ascend/ascend-toolkit
|
||||
ARG SGLANG_KERNEL_NPU_TAG=main
|
||||
|
||||
ARG PIP_INSTALL="python3 -m pip install --no-cache-dir"
|
||||
ARG DEVICE_TYPE
|
||||
|
||||
RUN if [ "$TARGETARCH" = "amd64" ]; then \
|
||||
echo "Using x86_64 dependencies"; \
|
||||
echo "PTA_URL=$PTA_URL_AMD64" >> /etc/environment_new; \
|
||||
elif [ "$TARGETARCH" = "arm64" ]; then \
|
||||
echo "Using aarch64 dependencies"; \
|
||||
echo "PTA_URL=$PTA_URL_ARM64" >> /etc/environment_new; \
|
||||
else \
|
||||
echo "Unsupported TARGETARCH: $TARGETARCH"; exit 1; \
|
||||
fi
|
||||
|
||||
WORKDIR /workspace
|
||||
|
||||
# Define environments
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
RUN pip config set global.index-url $PIP_INDEX_URL
|
||||
RUN if [ -n "$APTMIRROR" ];then sed -i "s|.*.ubuntu.com|$APTMIRROR|g" /etc/apt/sources.list ;fi
|
||||
|
||||
# Install development tools and utilities
|
||||
RUN apt-get update -y && apt upgrade -y && apt-get install -y \
|
||||
unzip \
|
||||
build-essential \
|
||||
cmake \
|
||||
vim \
|
||||
wget \
|
||||
curl \
|
||||
net-tools \
|
||||
zlib1g-dev \
|
||||
lld \
|
||||
clang \
|
||||
locales \
|
||||
ccache \
|
||||
openssl \
|
||||
libssl-dev \
|
||||
pkg-config \
|
||||
libgl1-mesa-glx \
|
||||
libgl1-mesa-dri \
|
||||
ca-certificates \
|
||||
&& rm -rf /var/cache/apt/* \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& update-ca-certificates \
|
||||
&& locale-gen en_US.UTF-8
|
||||
|
||||
ENV LANG=en_US.UTF-8
|
||||
ENV LANGUAGE=en_US:en
|
||||
ENV LC_ALL=en_US.UTF-8
|
||||
|
||||
|
||||
### Install MemFabric
|
||||
RUN ${PIP_INSTALL} memfabric-hybrid==1.0.8
|
||||
|
||||
### Install zbal
|
||||
RUN if [ "$TARGETARCH" = "arm64" ]; then \
|
||||
${PIP_INSTALL} memfabric-zbal==1.1.1; \
|
||||
fi
|
||||
|
||||
### Install SGLang Model Gateway
|
||||
RUN ${PIP_INSTALL} sglang-router
|
||||
|
||||
|
||||
### Install PyTorch and PTA
|
||||
RUN . /etc/environment_new && \
|
||||
(${PIP_INSTALL} torch==${PYTORCH_VERSION} torchvision==${TORCHVISION_VERSION} torchaudio==${TORCHAUDIO_VERSION} --index-url https://download.pytorch.org/whl/cpu) \
|
||||
&& (${PIP_INSTALL} ${PTA_URL})
|
||||
|
||||
|
||||
## Install triton-ascend
|
||||
RUN (${PIP_INSTALL} pybind11) && \
|
||||
(${PIP_INSTALL} triton-ascend==3.2.1.dev20260530 --extra-index-url=https://mirrors.huaweicloud.com/ascend/repos/pypi/nightly --trusted-host triton-ascend.osinfra.cn)
|
||||
|
||||
# Install SGLang (editable mode to preserve source and git history)
|
||||
RUN git clone https://github.com/sgl-project/sglang --branch $SGLANG_TAG /sgl-workspace/sglang && \
|
||||
cd /sgl-workspace/sglang/python && rm -rf pyproject.toml && mv pyproject_npu.toml pyproject.toml && \
|
||||
${PIP_INSTALL} -v -e .[all_npu]
|
||||
|
||||
# Install Deep-ep
|
||||
# pin wheel to 0.45.1 ref: https://github.com/pypa/wheel/issues/662
|
||||
RUN ${PIP_INSTALL} wheel==0.45.1 pybind11 pyyaml decorator scipy attrs psutil \
|
||||
&& mkdir sgl-kernel-npu \
|
||||
&& cd sgl-kernel-npu \
|
||||
&& wget https://github.com/sgl-project/sgl-kernel-npu/releases/download/${SGLANG_KERNEL_NPU_TAG}/sgl-kernel-npu-${SGLANG_KERNEL_NPU_TAG}-torch2.10.0-py311-cann${CANN_VERSION}-${DEVICE_TYPE}-$(arch).zip \
|
||||
&& unzip sgl-kernel-npu-${SGLANG_KERNEL_NPU_TAG}-torch2.10.0-py311-cann${CANN_VERSION}-${DEVICE_TYPE}-$(arch).zip \
|
||||
&& ${PIP_INSTALL} deep_ep*.whl sgl_kernel_npu*.whl \
|
||||
&& cd .. && rm -rf sgl-kernel-npu \
|
||||
&& cd "$(python3 -m pip show deep-ep | awk '/^Location:/ {print $2}')" && ln -sf deep_ep/deep_ep_cpp*.so
|
||||
|
||||
CMD ["/bin/bash"]
|
||||
@@ -0,0 +1,679 @@
|
||||
# Usage (to build SGLang ROCm docker image):
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx942 -t v0.5.10.post1-rocm700-mi30x -f rocm.Dockerfile .
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx942-rocm720 -t v0.5.10.post1-rocm720-mi30x -f rocm.Dockerfile .
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx950 -t v0.5.10.post1-rocm700-mi35x -f rocm.Dockerfile .
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx950-rocm720 -t v0.5.10.post1-rocm720-mi35x -f rocm.Dockerfile .
|
||||
|
||||
# Usage (to build SGLang ROCm + Mori docker image):
|
||||
# remove --build-arg NIC_BACKEND=ainic since new MoRI JIT will do NIC auto detection on target
|
||||
# Keep the build-arg for user to select the desired nic support, current choice: [ainic, bxnt]
|
||||
# if no set this arg, it will support nic auto detection. On a target with more than 1 type of
|
||||
# RDMA NICs installed (rare), overwrite w. runtime env MORI_DEVICE_NIC = "bnxt"|"ionic"|"mlx5"
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx942 --build-arg ENABLE_MORI=1 -t v0.5.10.post1-rocm700-mi30x -f rocm.Dockerfile .
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx942-rocm720 --build-arg ENABLE_MORI=1 -t v0.5.10.post1-rocm720-mi30x -f rocm.Dockerfile .
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx950 --build-arg ENABLE_MORI=1 -t v0.5.10.post1-rocm700-mi35x -f rocm.Dockerfile .
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx950-rocm720 --build-arg ENABLE_MORI=1 -t v0.5.10.post1-rocm720-mi35x -f rocm.Dockerfile .
|
||||
|
||||
# Usage (to build SGLang ROCm + NIXL docker image, for prefill/decode disaggregation):
|
||||
# Builds UCX (--with-rocm) and upstream ai-dynamo/nixl from source by default.
|
||||
# Set ENABLE_NIXL=0 to skip NIXL.
|
||||
# At runtime use --disaggregation-transfer-backend nixl (env is wired via /etc/bash.bashrc).
|
||||
# docker build --build-arg SGL_BRANCH=v0.5.10.post1 --build-arg GPU_ARCH=gfx950-rocm720 -t v0.5.10.post1-rocm720-mi35x -f rocm.Dockerfile .
|
||||
|
||||
# Default base images
|
||||
ARG BASE_IMAGE_942="rocm/sgl-dev:rocm7-vllm-20250904"
|
||||
ARG BASE_IMAGE_942_ROCM720="rocm/pytorch:rocm7.2_ubuntu22.04_py3.10_pytorch_release_2.9.1"
|
||||
ARG BASE_IMAGE_950="rocm/sgl-dev:rocm7-vllm-20250904"
|
||||
ARG BASE_IMAGE_950_ROCM720="rocm/pytorch:rocm7.2_ubuntu22.04_py3.10_pytorch_release_2.9.1"
|
||||
|
||||
# This is necessary for scope purpose
|
||||
ARG GPU_ARCH=gfx950
|
||||
|
||||
# ===============================
|
||||
# Base image 942 with rocm700 and args
|
||||
FROM $BASE_IMAGE_942 AS gfx942
|
||||
ENV BUILD_VLLM="0"
|
||||
ENV BUILD_TRITON="0"
|
||||
ENV BUILD_LLVM="0"
|
||||
ENV BUILD_AITER_ALL="1"
|
||||
ENV BUILD_MOONCAKE="1"
|
||||
ENV AITER_COMMIT_DEFAULT="9127c94a18e4398e1eba91f6639e910f0994ad02"
|
||||
|
||||
# ===============================
|
||||
# Base image 942 with rocm720 and args
|
||||
FROM $BASE_IMAGE_942_ROCM720 AS gfx942-rocm720
|
||||
ENV BUILD_VLLM="0"
|
||||
ENV BUILD_TRITON="1"
|
||||
ENV BUILD_LLVM="0"
|
||||
ENV BUILD_AITER_ALL="1"
|
||||
ENV BUILD_MOONCAKE="1"
|
||||
ENV AITER_COMMIT_DEFAULT="9127c94a18e4398e1eba91f6639e910f0994ad02"
|
||||
|
||||
# ===============================
|
||||
# Base image 950 and args
|
||||
FROM $BASE_IMAGE_950 AS gfx950
|
||||
ENV BUILD_VLLM="0"
|
||||
ENV BUILD_TRITON="0"
|
||||
ENV BUILD_LLVM="0"
|
||||
ENV BUILD_AITER_ALL="1"
|
||||
ENV BUILD_MOONCAKE="1"
|
||||
ENV AITER_COMMIT_DEFAULT="9127c94a18e4398e1eba91f6639e910f0994ad02"
|
||||
|
||||
# ===============================
|
||||
# Base image 950 with rocm720 and args
|
||||
FROM $BASE_IMAGE_950_ROCM720 AS gfx950-rocm720
|
||||
ENV BUILD_VLLM="0"
|
||||
ENV BUILD_TRITON="1"
|
||||
ENV BUILD_LLVM="0"
|
||||
ENV BUILD_AITER_ALL="1"
|
||||
ENV BUILD_MOONCAKE="1"
|
||||
ENV AITER_COMMIT_DEFAULT="9127c94a18e4398e1eba91f6639e910f0994ad02"
|
||||
|
||||
# ===============================
|
||||
# Chosen arch and args
|
||||
FROM ${GPU_ARCH}
|
||||
|
||||
# This is necessary for scope purpose, again
|
||||
ARG GPU_ARCH=gfx950
|
||||
ENV GPU_ARCH_LIST=${GPU_ARCH%-*}
|
||||
ENV PYTORCH_ROCM_ARCH=gfx942;gfx950
|
||||
|
||||
ARG SGL_REPO="https://github.com/sgl-project/sglang.git"
|
||||
ARG SGL_DEFAULT="main"
|
||||
ARG SGL_BRANCH=${SGL_DEFAULT}
|
||||
|
||||
# Version override for setuptools_scm (used in nightly builds)
|
||||
ARG SETUPTOOLS_SCM_PRETEND_VERSION=""
|
||||
|
||||
ARG TRITON_REPO="https://github.com/triton-lang/triton.git"
|
||||
ARG TRITON_COMMIT="42270451990532c67e69d753fbd026f28fcc4840"
|
||||
|
||||
ARG AITER_REPO="https://github.com/ROCm/aiter.git"
|
||||
ARG AITER_COMMIT=""
|
||||
ENV AITER_COMMIT="${AITER_COMMIT:-${AITER_COMMIT_DEFAULT}}"
|
||||
|
||||
ARG LLVM_REPO="https://github.com/jrbyrnes/llvm-project.git"
|
||||
ARG LLVM_BRANCH="MainOpSelV2"
|
||||
ARG LLVM_COMMIT="6520ace8227ffe2728148d5f3b9872a870b0a560"
|
||||
|
||||
ARG MOONCAKE_REPO="https://github.com/kvcache-ai/Mooncake.git"
|
||||
ARG MOONCAKE_COMMIT="01d1eb2a7ec37fd5e20a88573e9b4956e7846e9a"
|
||||
|
||||
ARG TILELANG_REPO="https://github.com/tile-ai/tilelang.git"
|
||||
ARG TILELANG_COMMIT="a55a82302bf7f3c5af635b5c9146f728185cc900"
|
||||
|
||||
ARG FHT_REPO="https://github.com/jeffdaily/fast-hadamard-transform.git"
|
||||
ARG FHT_BRANCH="rocm"
|
||||
ARG FHT_COMMIT="46efb7d776d38638fc39f3c803eaee3dd7016bd1"
|
||||
|
||||
ARG ENABLE_MORI=0
|
||||
ARG NIC_BACKEND=none
|
||||
|
||||
ARG MORI_REPO="https://github.com/ROCm/mori.git"
|
||||
ARG MORI_COMMIT="e31d426a13e96e1cbff96a1c904d291aefe8c46a"
|
||||
|
||||
# NIXL (upstream ai-dynamo/nixl) — KV transfer backend for prefill/decode disaggregation.
|
||||
# Built from source for ROCm; needs UCX built --with-rocm (built here from openucx).
|
||||
# Enabled by default; disable with --build-arg ENABLE_NIXL=0.
|
||||
ARG ENABLE_NIXL=1
|
||||
ARG UCX_REPO="https://github.com/openucx/ucx.git"
|
||||
ARG UCX_BRANCH="v1.19.x"
|
||||
ARG NIXL_REPO="https://github.com/ai-dynamo/nixl.git"
|
||||
ARG NIXL_COMMIT="c28061f9782e099f975bcc79198b7b5a1a36cc40"
|
||||
|
||||
# AMD AINIC apt repo settings
|
||||
ARG AINIC_VERSION=1.117.5-a-38
|
||||
ARG UBUNTU_CODENAME=jammy
|
||||
|
||||
# Optional Ubuntu mirror override + apt hardening.
|
||||
# - UBUNTU_MIRROR is empty by default (no behaviour change for local builds).
|
||||
# When set (typically in CI), all http://*archive.ubuntu.com and
|
||||
# http://*security.ubuntu.com entries in /etc/apt/sources.list are rewritten
|
||||
# to point at the given base URL, e.g.
|
||||
# --build-arg UBUNTU_MIRROR=https://archive.ubuntu.com
|
||||
# --build-arg UBUNTU_MIRROR=https://tw.archive.ubuntu.com
|
||||
# --build-arg UBUNTU_MIRROR=http://internal-cache.example.com
|
||||
# This mirrors the pattern already used in docker/Dockerfile (NVIDIA) and
|
||||
# docker/npu.Dockerfile, and lets CI runners that cannot reach Canonical's
|
||||
# port-80 mirror IPs still complete `apt-get update`.
|
||||
# - The 80-net-hardening apt config adds retries + per-request timeout so that
|
||||
# transient mirror flakes don't immediately fail a build (apt's default is 0
|
||||
# retries).
|
||||
ARG UBUNTU_MIRROR=
|
||||
USER root
|
||||
|
||||
RUN if [ -n "$UBUNTU_MIRROR" ]; then \
|
||||
sed -i "s|http://[^[:space:]/]*archive.ubuntu.com|$UBUNTU_MIRROR|g" /etc/apt/sources.list && \
|
||||
sed -i "s|http://[^[:space:]/]*security.ubuntu.com|$UBUNTU_MIRROR|g" /etc/apt/sources.list; \
|
||||
fi && \
|
||||
printf 'Acquire::Retries "5";\nAcquire::http::Timeout "30";\nAcquire::https::Timeout "30";\n' \
|
||||
> /etc/apt/apt.conf.d/80-net-hardening
|
||||
|
||||
# Fix hipDeviceGetName returning empty string in ROCm 7.0 docker images.
|
||||
# The ROCm 7.0 base image is missing libdrm-amdgpu-common which provides the
|
||||
# amdgpu.ids device-ID-to-marketing-name mapping file.
|
||||
# ROCm 7.2 base images already ship these packages, so this step is skipped.
|
||||
# See https://github.com/ROCm/ROCm/issues/5992
|
||||
RUN set -eux; \
|
||||
case "${GPU_ARCH}" in \
|
||||
*rocm720*) \
|
||||
echo "ROCm 7.2 (GPU_ARCH=${GPU_ARCH}): libdrm-amdgpu packages already present, skipping"; \
|
||||
;; \
|
||||
*) \
|
||||
echo "ROCm 7.0 (GPU_ARCH=${GPU_ARCH}): installing libdrm-amdgpu packages"; \
|
||||
curl -fsSL https://repo.radeon.com/rocm/rocm.gpg.key \
|
||||
| gpg --dearmor -o /etc/apt/keyrings/amdgpu-graphics.gpg \
|
||||
&& echo 'deb [arch=amd64,i386 signed-by=/etc/apt/keyrings/amdgpu-graphics.gpg] https://repo.radeon.com/graphics/7.0/ubuntu jammy main' \
|
||||
> /etc/apt/sources.list.d/amdgpu-graphics.list \
|
||||
&& apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
libdrm-amdgpu-common \
|
||||
libdrm-amdgpu-amdgpu1 \
|
||||
libdrm2-amdgpu \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& cp /opt/amdgpu/share/libdrm/amdgpu.ids /usr/share/libdrm/amdgpu.ids; \
|
||||
;; \
|
||||
esac
|
||||
|
||||
|
||||
# Install some basic utilities
|
||||
RUN python -m pip install --upgrade pip && pip install setuptools_scm
|
||||
RUN apt-get purge -y sccache; python -m pip uninstall -y sccache; rm -f "$(which sccache)"
|
||||
|
||||
# Install AMD SMI Python package from ROCm distribution.
|
||||
# The ROCm 7.2 base image (rocm/pytorch) does not pre-install this package.
|
||||
RUN set -eux; \
|
||||
case "${GPU_ARCH}" in \
|
||||
*rocm720*) \
|
||||
echo "ROCm 7.2 flavor detected from GPU_ARCH=${GPU_ARCH}"; \
|
||||
cd /opt/rocm/share/amd_smi \
|
||||
&& python3 -m pip install --no-cache-dir . \
|
||||
;; \
|
||||
*) \
|
||||
echo "Not rocm720 (GPU_ARCH=${GPU_ARCH}), skip amdsmi installation"; \
|
||||
;; \
|
||||
esac
|
||||
|
||||
WORKDIR /sgl-workspace
|
||||
|
||||
# -----------------------
|
||||
# llvm
|
||||
RUN if [ "$BUILD_LLVM" = "1" ]; then \
|
||||
ENV HIP_CLANG_PATH="/sgl-workspace/llvm-project/build/bin/" \
|
||||
git clone --single-branch ${LLVM_REPO} -b ${LLVM_BRANCH} \
|
||||
&& cd llvm-project \
|
||||
&& git checkout ${LLVM_COMMIT} \
|
||||
&& mkdir build \
|
||||
&& cd build \
|
||||
&& cmake -DCMAKE_BUILD_TYPE=Release -DLLVM_ENABLE_ASSERTIONS=1 -DLLVM_TARGETS_TO_BUILD="AMDGPU;X86" -DLLVM_ENABLE_PROJECTS="clang;lld;" -DLLVM_ENABLE_RUNTIMES="compiler-rt" ../llvm \
|
||||
&& make -j$(nproc); \
|
||||
fi
|
||||
|
||||
# -----------------------
|
||||
# AITER
|
||||
# Unset setuptools_scm override so AITER gets its own version (AITER_COMMIT), not SGLang's
|
||||
# (SETUPTOOLS_SCM_PRETEND_VERSION is set later for SGLang nightly builds and would otherwise
|
||||
# leak into AITER's version when AITER uses setuptools_scm)
|
||||
|
||||
ENV SETUPTOOLS_SCM_PRETEND_VERSION=
|
||||
# Keep the base image's Torch-compatible Triton by default. Override with
|
||||
# AITER_USE_SYSTEM_TRITON=0 when intentionally testing aiter-managed Triton.
|
||||
ENV AITER_USE_SYSTEM_TRITON=1
|
||||
RUN pip uninstall -y aiter
|
||||
# Use `checkout -f` so the smudge-filter-induced "dirty" working tree from
|
||||
# AITER's .gitattributes (*.csv text eol=lf, added in ROCm/aiter#3370) does not
|
||||
# block switching to commits that predate that rule (e.g. the current default
|
||||
# AITER_COMMIT_DEFAULT). The working tree was just produced by a fresh
|
||||
# `git clone` above, so there are no real user changes to preserve.
|
||||
RUN git clone ${AITER_REPO} \
|
||||
&& cd aiter \
|
||||
&& git checkout -f ${AITER_COMMIT} \
|
||||
&& git submodule update --init --recursive \
|
||||
&& pip install -r requirements.txt
|
||||
|
||||
RUN cd aiter \
|
||||
&& echo "[AITER] GPU_ARCH=${GPU_ARCH}" \
|
||||
&& echo "[AITER] AITER_USE_SYSTEM_TRITON=${AITER_USE_SYSTEM_TRITON}" \
|
||||
&& if [ "$BUILD_AITER_ALL" = "1" ] && [ "$BUILD_LLVM" = "1" ]; then \
|
||||
sh -c "HIP_CLANG_PATH=/sgl-workspace/llvm-project/build/bin/ PREBUILD_KERNELS=1 GPU_ARCHS=$GPU_ARCH_LIST python setup.py build_ext --inplace" \
|
||||
&& sh -c "HIP_CLANG_PATH=/sgl-workspace/llvm-project/build/bin/ GPU_ARCHS=$GPU_ARCH_LIST pip install --config-settings editable_mode=compat -e ."; \
|
||||
elif [ "$BUILD_AITER_ALL" = "1" ]; then \
|
||||
sh -c "PREBUILD_KERNELS=1 GPU_ARCHS=$GPU_ARCH_LIST python setup.py build_ext --inplace" \
|
||||
&& sh -c "GPU_ARCHS=$GPU_ARCH_LIST pip install --config-settings editable_mode=compat -e ."; \
|
||||
else \
|
||||
sh -c "GPU_ARCHS=$GPU_ARCH_LIST pip install --config-settings editable_mode=compat -e ."; \
|
||||
fi \
|
||||
&& echo "export PYTHONPATH=/sgl-workspace/aiter:\${PYTHONPATH}" >> /etc/bash.bashrc
|
||||
|
||||
# -----------------------
|
||||
# Build Mooncake
|
||||
ENV PATH=$PATH:/usr/local/go/bin
|
||||
|
||||
RUN if [ "$BUILD_MOONCAKE" = "1" ]; then \
|
||||
apt update && apt install -y zip unzip wget && \
|
||||
apt install -y gcc make libtool autoconf librdmacm-dev rdmacm-utils infiniband-diags ibverbs-utils perftest ethtool libibverbs-dev rdma-core && \
|
||||
apt install -y openssh-server openmpi-bin openmpi-common libopenmpi-dev && \
|
||||
git clone ${MOONCAKE_REPO} && \
|
||||
cd Mooncake && \
|
||||
git checkout ${MOONCAKE_COMMIT} && \
|
||||
git submodule update --init --recursive && \
|
||||
bash dependencies.sh -y && \
|
||||
rm -rf /usr/local/go && \
|
||||
wget https://go.dev/dl/go1.22.2.linux-amd64.tar.gz && \
|
||||
tar -C /usr/local -xzf go1.22.2.linux-amd64.tar.gz && \
|
||||
rm go1.22.2.linux-amd64.tar.gz && \
|
||||
mkdir -p build && \
|
||||
cd build && \
|
||||
cmake .. -DUSE_HIP=ON -DUSE_ETCD=ON -DENABLE_MULTI_PROTOCOL=ON -DWITH_STORE=ON -DBUILD_UNIT_TESTS=OFF && \
|
||||
make -j "$(nproc)" && make install; \
|
||||
fi
|
||||
|
||||
# -----------------------
|
||||
# Build SGLang
|
||||
ARG BUILD_TYPE=all
|
||||
|
||||
# Set version for setuptools_scm if provided (for nightly builds). Only pass in the SGLang
|
||||
# pip install RUN so it does not affect AITER, sgl-model-gateway, TileLang, FHT, MORI, etc.
|
||||
ARG SETUPTOOLS_SCM_PRETEND_VERSION
|
||||
|
||||
RUN pip install IPython \
|
||||
&& pip install orjson \
|
||||
&& pip install python-multipart \
|
||||
&& pip install torchao==0.9.0 \
|
||||
&& pip install pybind11
|
||||
|
||||
RUN pip uninstall -y sgl_kernel sglang
|
||||
RUN git clone ${SGL_REPO} \
|
||||
&& cd sglang \
|
||||
&& if [ "${SGL_BRANCH}" = ${SGL_DEFAULT} ]; then \
|
||||
echo "Using ${SGL_DEFAULT}, default branch."; \
|
||||
git checkout ${SGL_DEFAULT}; \
|
||||
else \
|
||||
echo "Using ${SGL_BRANCH} branch."; \
|
||||
git checkout ${SGL_BRANCH}; \
|
||||
fi \
|
||||
&& cd sgl-kernel \
|
||||
&& rm -f pyproject.toml \
|
||||
&& mv pyproject_rocm.toml pyproject.toml \
|
||||
&& AMDGPU_TARGET=$GPU_ARCH_LIST python setup_rocm.py install \
|
||||
&& cd .. \
|
||||
&& rm -rf python/pyproject.toml && mv python/pyproject_other.toml python/pyproject.toml \
|
||||
&& if [ "$BUILD_TYPE" = "srt" ]; then \
|
||||
export SETUPTOOLS_SCM_PRETEND_VERSION="${SETUPTOOLS_SCM_PRETEND_VERSION}" && python -m pip --no-cache-dir install -e "python[srt_hip,diffusion_hip]"; \
|
||||
else \
|
||||
export SETUPTOOLS_SCM_PRETEND_VERSION="${SETUPTOOLS_SCM_PRETEND_VERSION}" && python -m pip --no-cache-dir install -e "python[all_hip]"; \
|
||||
fi
|
||||
|
||||
RUN python -m pip cache purge
|
||||
|
||||
# Copy config files to support MI300X in virtualized environments (MI300X_VF). Symlinks will not be created in image build.
|
||||
RUN find /sgl-workspace/sglang/python/sglang/srt/layers/quantization/configs/ \
|
||||
/sgl-workspace/sglang/python/sglang/srt/layers/moe/fused_moe_triton/configs/ \
|
||||
-type f -name '*MI300X*' | xargs -I {} sh -c 'vf_config=$(echo "$1" | sed "s/MI300X/MI300X_VF/"); cp "$1" "$vf_config"' -- {}
|
||||
|
||||
# Install Rust toolchain for sgl-model-gateway
|
||||
ENV PATH="/root/.cargo/bin:${PATH}"
|
||||
RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y \
|
||||
&& rustc --version && cargo --version
|
||||
ENV CARGO_BUILD_JOBS=4
|
||||
|
||||
# Build and install sgl-model-gateway
|
||||
RUN python3 -m pip install --no-cache-dir "maturin<1.14" \
|
||||
&& sed -i -E 's|^(smg-[a-zA-Z-]+)\s*=\s*"~1\.0\.0"|\1 = "=1.0.0"|' \
|
||||
/sgl-workspace/sglang/sgl-model-gateway/Cargo.toml \
|
||||
&& grep -E '^smg-' /sgl-workspace/sglang/sgl-model-gateway/Cargo.toml \
|
||||
&& cd /sgl-workspace/sglang/sgl-model-gateway/bindings/python \
|
||||
&& ulimit -n 65536 && maturin build --release --features vendored-openssl --out dist \
|
||||
&& python3 -m pip install --force-reinstall dist/*.whl \
|
||||
&& rm -rf /root/.cache
|
||||
|
||||
# -----------------------
|
||||
# TileLang
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
ENV LIBGL_ALWAYS_INDIRECT=1
|
||||
RUN echo "LC_ALL=en_US.UTF-8" >> /etc/environment
|
||||
|
||||
RUN /bin/bash -lc 'set -euo pipefail; \
|
||||
echo "[TileLang] Building TileLang for ${GPU_ARCH}"; \
|
||||
# System dependencies (NO llvm-dev to avoid llvm-config-16 shadowing)
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential git wget curl ca-certificates gnupg \
|
||||
libgtest-dev libgmock-dev \
|
||||
libprotobuf-dev protobuf-compiler libgflags-dev libsqlite3-dev \
|
||||
python3 python3-dev python3-setuptools python3-pip python3-apt \
|
||||
gcc libtinfo-dev zlib1g-dev libedit-dev libxml2-dev vim \
|
||||
cmake ninja-build pkg-config libstdc++6 software-properties-common \
|
||||
&& rm -rf /var/lib/apt/lists/*; \
|
||||
\
|
||||
# Prefer the container venv
|
||||
VENV_PY="/opt/venv/bin/python"; \
|
||||
VENV_PIP="/opt/venv/bin/pip"; \
|
||||
if [ ! -x "$VENV_PY" ]; then VENV_PY="python3"; fi; \
|
||||
if [ ! -x "$VENV_PIP" ]; then VENV_PIP="pip3"; fi; \
|
||||
\
|
||||
# Build GoogleTest static libs (Ubuntu package ships sources only)
|
||||
cmake -S /usr/src/googletest -B /tmp/build-gtest -DBUILD_GTEST=ON -DBUILD_GMOCK=ON -DCMAKE_BUILD_TYPE=Release && \
|
||||
cmake --build /tmp/build-gtest -j"$(nproc)" && \
|
||||
cp -v /tmp/build-gtest/lib/*.a /usr/lib/x86_64-linux-gnu/ && \
|
||||
rm -rf /tmp/build-gtest; \
|
||||
\
|
||||
# Keep setuptools < 80 (compat with base image). Pin cmake to the last known-good
|
||||
# 4.3.4: cmake 4.4's gtest_discover_tests breaks the (pinned) MoRI build with a
|
||||
# JSON parse error. This image is rebuilt daily, so pin the exact version for
|
||||
# reproducible builds rather than letting cmake drift.
|
||||
"$VENV_PIP" install --upgrade "setuptools>=77.0.3,<80" wheel "cmake==4.3.4" ninja scikit-build-core && \
|
||||
"$VENV_PIP" cache purge || true; \
|
||||
\
|
||||
# Locate ROCm llvm-config; fallback to installing LLVM 18 if missing
|
||||
LLVM_CONFIG_PATH=""; \
|
||||
for p in /opt/rocm/llvm/bin/llvm-config /opt/rocm/llvm-*/bin/llvm-config /opt/rocm-*/llvm*/bin/llvm-config; do \
|
||||
if [ -x "$p" ]; then LLVM_CONFIG_PATH="$p"; break; fi; \
|
||||
done; \
|
||||
if [ -z "$LLVM_CONFIG_PATH" ]; then \
|
||||
echo "[TileLang] ROCm llvm-config not found; installing LLVM 18..."; \
|
||||
curl -fsSL https://apt.llvm.org/llvm-snapshot.gpg.key | gpg --dearmor -o /etc/apt/keyrings/llvm.gpg; \
|
||||
echo "deb [signed-by=/etc/apt/keyrings/llvm.gpg] http://apt.llvm.org/jammy/ llvm-toolchain-jammy-18 main" > /etc/apt/sources.list.d/llvm.list; \
|
||||
apt-get update; \
|
||||
apt-get install -y --no-install-recommends llvm-18; \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
LLVM_CONFIG_PATH="$(command -v llvm-config-18)"; \
|
||||
if [ -z "$LLVM_CONFIG_PATH" ]; then echo "ERROR: llvm-config-18 not found after install"; exit 1; fi; \
|
||||
fi; \
|
||||
echo "[TileLang] Using LLVM_CONFIG at: $LLVM_CONFIG_PATH"; \
|
||||
export PATH="$(dirname "$LLVM_CONFIG_PATH"):/usr/local/bin:${PATH}"; \
|
||||
export LLVM_CONFIG="$LLVM_CONFIG_PATH"; \
|
||||
\
|
||||
# Optional shim for tools that expect llvm-config-16
|
||||
mkdir -p /usr/local/bin && \
|
||||
printf "#!/usr/bin/env bash\nexec \"%s\" \"\$@\"\n" "$LLVM_CONFIG_PATH" > /usr/local/bin/llvm-config-16 && \
|
||||
chmod +x /usr/local/bin/llvm-config-16; \
|
||||
\
|
||||
# TVM Python bits need Cython + z3 before configure.
|
||||
# Pin z3-solver==4.15.4.0: 4.15.4.0 has a manylinux wheel; 4.15.5.0 has no wheel and builds from source (fails: C++20 <format> needs GCC 14+, image has GCC 11).
|
||||
"$VENV_PIP" install --no-cache-dir "cython>=0.29.36,<3.0" "apache-tvm-ffi @ git+https://github.com/apache/tvm-ffi.git@37d0485b2058885bf4e7a486f7d7b2174a8ac1ce" "z3-solver==4.15.4.0"; \
|
||||
\
|
||||
# Clone + pin TileLang (bundled TVM), then build
|
||||
git clone --recursive "${TILELANG_REPO}" /opt/tilelang && \
|
||||
cd /opt/tilelang && \
|
||||
git fetch --depth=1 origin "${TILELANG_COMMIT}" || true && \
|
||||
git checkout -f "${TILELANG_COMMIT}" && \
|
||||
git submodule update --init --recursive && \
|
||||
export CMAKE_ARGS="-DUSE_CUDA=OFF -DUSE_ROCM=ON -DROCM_PATH=/opt/rocm -DLLVM_CONFIG=${LLVM_CONFIG} -DSKBUILD_SABI_VERSION= ${CMAKE_ARGS:-}" && \
|
||||
"$VENV_PIP" install -e . -v --no-build-isolation --no-deps; \
|
||||
if [ -f pyproject.toml ]; then sed -i "/^[[:space:]]*\"torch/d" pyproject.toml || true; fi; \
|
||||
"$VENV_PIP" cache purge || true; \
|
||||
"$VENV_PY" -c "import tilelang; print(tilelang.__version__)"'
|
||||
|
||||
# -----------------------
|
||||
# Hadamard-transform (HIP build)
|
||||
RUN /bin/bash -lc 'set -euo pipefail; \
|
||||
git clone --branch "${FHT_BRANCH}" "${FHT_REPO}" fast-hadamard-transform; \
|
||||
cd fast-hadamard-transform; \
|
||||
git checkout -f "${FHT_COMMIT}"; \
|
||||
python setup.py install'
|
||||
|
||||
# -----------------------
|
||||
# Python tools
|
||||
RUN python3 -m pip install --no-cache-dir \
|
||||
py-spy \
|
||||
pre-commit \
|
||||
tabulate
|
||||
|
||||
# -----------------------
|
||||
# MORI (optional)
|
||||
RUN /bin/bash -lc 'set -euo pipefail; \
|
||||
if [ "${ENABLE_MORI}" != "1" ]; then \
|
||||
echo "[MORI] Skipping (ENABLE_MORI=${ENABLE_MORI})"; \
|
||||
exit 0; \
|
||||
fi; \
|
||||
echo "[MORI] Enabling MORI (NIC_BACKEND=${NIC_BACKEND})"; \
|
||||
\
|
||||
# Base deps for MORI build
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential \
|
||||
g++ \
|
||||
jq \
|
||||
libopenmpi-dev \
|
||||
libpci-dev \
|
||||
initramfs-tools \
|
||||
&& rm -rf /var/lib/apt/lists/*; \
|
||||
\
|
||||
# NIC backend deps — mori auto-detects NIC at runtime (MORI_DEVICE_NIC env var override).
|
||||
# Only vendor packages are installed here for dlopen (e.g. libionic.so); no compile-time flags needed.
|
||||
case "${NIC_BACKEND}" in \
|
||||
# default: install ainic and bxnt driver
|
||||
none) \
|
||||
apt-get update && apt-get install -y --no-install-recommends ca-certificates curl gnupg apt-transport-https && \
|
||||
rm -rf /var/lib/apt/lists/* && mkdir -p /etc/apt/keyrings; \
|
||||
curl -fsSL https://repo.radeon.com/rocm/rocm.gpg.key | gpg --dearmor > /etc/apt/keyrings/amdainic.gpg; \
|
||||
echo "deb [arch=amd64 signed-by=/etc/apt/keyrings/amdainic.gpg] https://repo.radeon.com/amdainic/pensando/ubuntu/${AINIC_VERSION} ${UBUNTU_CODENAME} main" \
|
||||
> /etc/apt/sources.list.d/amdainic.list; \
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
libionic-dev \
|
||||
ionic-common \
|
||||
; \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
install -m 0755 -d /etc/apt/keyrings \
|
||||
&& curl -fsSL https://packages.broadcom.com/artifactory/api/security/keypair/PackagesKey/public -o /etc/apt/keyrings/broadcom-nic.asc \
|
||||
&& chmod a+r /etc/apt/keyrings/broadcom-nic.asc \
|
||||
&& echo "deb [arch=amd64 signed-by=/etc/apt/keyrings/broadcom-nic.asc] https://packages.broadcom.com/artifactory/ethernet-nic-debian-public jammy main" > /etc/apt/sources.list.d/broadcom-nic.list \
|
||||
&& apt-get update \
|
||||
&& apt-get install -y ibverbs-utils bnxt-rocelib=235.2.86.0 \
|
||||
&& cp /usr/local/lib/x86_64-linux-gnu/libbnxt_re* /usr/local/lib/. \
|
||||
;; \
|
||||
# AMD NIC
|
||||
ainic) \
|
||||
apt-get update && apt-get install -y --no-install-recommends ca-certificates curl gnupg apt-transport-https && \
|
||||
rm -rf /var/lib/apt/lists/* && mkdir -p /etc/apt/keyrings; \
|
||||
curl -fsSL https://repo.radeon.com/rocm/rocm.gpg.key | gpg --dearmor > /etc/apt/keyrings/amdainic.gpg; \
|
||||
echo "deb [arch=amd64 signed-by=/etc/apt/keyrings/amdainic.gpg] https://repo.radeon.com/amdainic/pensando/ubuntu/${AINIC_VERSION} ${UBUNTU_CODENAME} main" \
|
||||
> /etc/apt/sources.list.d/amdainic.list; \
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
libionic-dev \
|
||||
ionic-common \
|
||||
; \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
;; \
|
||||
bnxt) \
|
||||
echo "[MORI] Enabling Broadcom BNXT backend"; \
|
||||
apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ca-certificates curl \
|
||||
&& install -m 0755 -d /etc/apt/keyrings \
|
||||
&& curl -fsSL https://packages.broadcom.com/artifactory/api/security/keypair/PackagesKey/public -o /etc/apt/keyrings/broadcom-nic.asc \
|
||||
&& chmod a+r /etc/apt/keyrings/broadcom-nic.asc \
|
||||
&& echo "deb [arch=amd64 signed-by=/etc/apt/keyrings/broadcom-nic.asc] https://packages.broadcom.com/artifactory/ethernet-nic-debian-public jammy main" > /etc/apt/sources.list.d/broadcom-nic.list \
|
||||
&& apt-get update \
|
||||
&& apt-get install -y ibverbs-utils bnxt-rocelib=235.2.86.0 \
|
||||
&& cp /usr/local/lib/x86_64-linux-gnu/libbnxt_re* /usr/local/lib/. \
|
||||
;; \
|
||||
*) \
|
||||
echo "ERROR: unknown NIC_BACKEND=${NIC_BACKEND}. Use one of: none, ainic"; \
|
||||
exit 2; \
|
||||
;; \
|
||||
esac; \
|
||||
\
|
||||
# Build/install MORI
|
||||
export MORI_GPU_ARCHS="${GPU_ARCH_LIST}"; \
|
||||
echo "[MORI] MORI_GPU_ARCHS=${MORI_GPU_ARCHS} NIC_BACKEND=${NIC_BACKEND}"; \
|
||||
rm -rf /sgl-workspace/mori; \
|
||||
git clone "${MORI_REPO}" /sgl-workspace/mori; \
|
||||
cd /sgl-workspace/mori; \
|
||||
git checkout "${MORI_COMMIT}"; \
|
||||
git submodule update --init --recursive; \
|
||||
python3 setup.py develop; \
|
||||
python3 -c "import os, torch; print(os.path.join(os.path.dirname(torch.__file__), \"lib\"))" > /etc/ld.so.conf.d/torch.conf; \
|
||||
ldconfig; \
|
||||
echo "export PYTHONPATH=/sgl-workspace/mori:\${PYTHONPATH}" >> /etc/bash.bashrc; \
|
||||
echo "[MORI] Done."'
|
||||
|
||||
# -----------------------
|
||||
# NIXL — upstream ai-dynamo/nixl KV transfer backend for PD disaggregation on ROCm.
|
||||
# Builds UCX (--with-rocm) + nixl from source by default; skip with ENABLE_NIXL=0.
|
||||
# --no-build-isolation reuses the image's ROCm torch (nixl pins torch==2.11.* as a build dep,
|
||||
# which would otherwise pull a multi-GB CUDA torch); --no-deps keeps CUDA runtime deps out.
|
||||
# wheel_variant=rocm names the pkg nixl_rocm, so symlink `nixl` since SGLang imports plain nixl.
|
||||
# taskflow (header-only) is provided via pkg-config so meson skips its broken upstream wrap
|
||||
# download (GitHub regenerated the v3.10.0 tarball, breaking the pinned source_hash).
|
||||
RUN /bin/bash -lc 'set -euo pipefail; \
|
||||
[ "${ENABLE_NIXL}" = "1" ] || { echo "[NIXL] skip (ENABLE_NIXL=${ENABLE_NIXL})"; exit 0; }; \
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential autoconf automake libtool pkg-config git \
|
||||
libibverbs-dev librdmacm-dev rdma-core && rm -rf /var/lib/apt/lists/*; \
|
||||
pip install --no-cache-dir meson ninja pybind11 meson-python patchelf pyyaml; \
|
||||
git clone --depth=1 -b "${UCX_BRANCH}" "${UCX_REPO}" /sgl-workspace/ucx; \
|
||||
cd /sgl-workspace/ucx && ./autogen.sh && mkdir build && cd build && \
|
||||
../configure --prefix=/opt/ucx --enable-shared --disable-static --disable-doxygen-doc \
|
||||
--enable-optimizations --enable-devel-headers \
|
||||
--with-rocm=/opt/rocm --with-verbs --with-dm --enable-mt && \
|
||||
make -j"$(nproc)" && make install; \
|
||||
git clone --depth=1 -b v3.10.0 https://github.com/taskflow/taskflow.git /sgl-workspace/taskflow; \
|
||||
cp -r /sgl-workspace/taskflow/taskflow /usr/local/include/; \
|
||||
mkdir -p /usr/local/lib/pkgconfig; \
|
||||
printf "Name: taskflow\nDescription: Taskflow\nVersion: 3.10.0\nCflags: -I/usr/local/include\n" > /usr/local/lib/pkgconfig/taskflow.pc; \
|
||||
git clone "${NIXL_REPO}" /sgl-workspace/nixl && cd /sgl-workspace/nixl && git checkout -f "${NIXL_COMMIT}"; \
|
||||
CXXFLAGS="-Wno-error" LD_LIBRARY_PATH="/opt/ucx/lib:/opt/rocm/lib" PKG_CONFIG_PATH="/usr/local/lib/pkgconfig" \
|
||||
pip install . --no-deps --no-build-isolation \
|
||||
--config-settings=setup-args="-Ducx_path=/opt/ucx" \
|
||||
--config-settings=setup-args="-Dwheel_variant=rocm" \
|
||||
--config-settings=setup-args="-Denable_plugins=UCX,POSIX"; \
|
||||
SITE=$(python3 -c "import sysconfig; print(sysconfig.get_paths()[\"purelib\"])"); \
|
||||
ln -sfn nixl_rocm "$SITE/nixl"; \
|
||||
echo "export LD_LIBRARY_PATH=/opt/ucx/lib:\${LD_LIBRARY_PATH}" >> /etc/bash.bashrc'
|
||||
|
||||
# -----------------------
|
||||
# Hot patch: torch-ROCm
|
||||
# The artifact hardcoded the supported triton version to be 3.5.1.
|
||||
# Rewrite the restriction directly.
|
||||
ARG TORCH_ROCM_FILE="torch-2.9.1+rocm7.2.0.lw.git7e1940d4-cp310-cp310-linux_x86_64.whl"
|
||||
RUN mkdir /tmp/whl && cd /tmp/whl \
|
||||
&& export TORCH_ROCM_FILE="${TORCH_ROCM_FILE}" \
|
||||
&& cat > hack.py <<"PY"
|
||||
import zipfile, csv, os, re
|
||||
from pathlib import Path
|
||||
|
||||
fname = os.environ["TORCH_ROCM_FILE"]
|
||||
in_whl = Path("/") / fname
|
||||
out_whl = Path("/tmp")/ fname
|
||||
work = Path("/tmp/whl")
|
||||
|
||||
# 1) Extract
|
||||
with zipfile.ZipFile(in_whl, "r") as z:
|
||||
z.extractall(work)
|
||||
|
||||
# 2) Locate dist-info and patch METADATA (edit this logic to match your exact line)
|
||||
dist_info = next(work.glob("*.dist-info"))
|
||||
meta = dist_info / "METADATA"
|
||||
txt = meta.read_text(encoding="utf-8")
|
||||
|
||||
# Example: replace one exact requirement form.
|
||||
# Adjust the string to match what you actually see.
|
||||
pat = r"^Requires-Dist:\s*triton==3.5.1[^\s]*;"
|
||||
txt2, n = re.subn(pat, r"triton>=3.5.1;", txt, flags=re.MULTILINE)
|
||||
if txt2 == txt:
|
||||
raise SystemExit("Did not find expected Requires-Dist line to replace in METADATA")
|
||||
meta.write_text(txt2, encoding="utf-8")
|
||||
|
||||
# 3) Hacky step: blank hash/size columns in RECORD
|
||||
record = dist_info / "RECORD"
|
||||
rows = []
|
||||
with record.open(newline="", encoding="utf-8") as f:
|
||||
for r in csv.reader(f):
|
||||
if not r:
|
||||
continue
|
||||
# keep filename, blank out hash and size
|
||||
rows.append([r[0], "", ""])
|
||||
with record.open("w", newline="", encoding="utf-8") as f:
|
||||
csv.writer(f).writerows(rows)
|
||||
|
||||
# 4) Re-zip as a wheel
|
||||
with zipfile.ZipFile(out_whl, "w", compression=zipfile.ZIP_DEFLATED) as z:
|
||||
for p in work.rglob("*"):
|
||||
if p.is_file():
|
||||
z.write(p, p.relative_to(work).as_posix())
|
||||
|
||||
print("Wrote", out_whl)
|
||||
PY
|
||||
|
||||
RUN cd /tmp/whl \
|
||||
&& case "${GPU_ARCH}" in \
|
||||
*rocm720*) \
|
||||
echo "ROCm 7.2 flavor detected from GPU_ARCH=${GPU_ARCH}"; \
|
||||
python hack.py \
|
||||
&& python3 -m pip install --force --no-deps /tmp/${TORCH_ROCM_FILE} \
|
||||
&& rm -fr /tmp/whl /tmp/${TORCH_ROCM_FILE} \
|
||||
;; \
|
||||
*) \
|
||||
echo "Not rocm720 (GPU_ARCH=${GPU_ARCH}), skip patch"; \
|
||||
;; \
|
||||
esac
|
||||
|
||||
|
||||
# -----------------------
|
||||
# Hot patch: Triton
|
||||
# For ROCm 7.2, this custom build breaks pip dependency management,
|
||||
# so future `pip install` will break the ROCm stack.
|
||||
# A workaround for this is to reinstall the default triton
|
||||
# wheel with the `rocm/pytorch` image in the root directory.
|
||||
RUN if [ "$BUILD_TRITON" = "1" ]; then \
|
||||
pip uninstall -y triton \
|
||||
&& apt install -y cmake \
|
||||
&& git clone ${TRITON_REPO} triton-custom \
|
||||
&& cd triton-custom \
|
||||
&& git checkout ${TRITON_COMMIT} \
|
||||
&& pip install -r python/requirements.txt \
|
||||
&& pip install -e . \
|
||||
&& if [ -d python/triton_kernels ]; then pip install -e python/triton_kernels --no-deps; fi; \
|
||||
fi
|
||||
|
||||
# -----------------------
|
||||
# Hot patch: transformers dynamic_module_utils symlink bug (v5.12.1).
|
||||
# _compute_local_source_files_hash calls Path(...).resolve() on custom-code
|
||||
# module files, following the HF-cache snapshots/<hash>/x.py -> blobs/<blob>
|
||||
# symlink. trust_remote_code models whose custom code uses relative imports
|
||||
# (e.g. Kimi-K2.6's kimi_k25_vision_processing.py: `from .media_utils import`)
|
||||
# then crash with FileNotFoundError: .../blobs/<name>.py at processor init.
|
||||
# Mirrors upstream transformers PR #46618 (merged, not yet released): drop the
|
||||
# .resolve() on the module file and its relative-import sources so the snapshot
|
||||
# .py names (not the blob targets) are used. Self-skips once transformers ships
|
||||
# the fix; fails the build loudly if the pattern is present but unpatched.
|
||||
RUN python3 - <<'PY'
|
||||
import pathlib
|
||||
import transformers.dynamic_module_utils as m
|
||||
|
||||
MARKS = ["Path(resolved_module_file).resolve()", "Path(source_file).resolve()"]
|
||||
path = pathlib.Path(m.__file__)
|
||||
src = path.read_text()
|
||||
if not any(mark in src for mark in MARKS):
|
||||
print("transformers dynamic_module_utils already fixed; no patch needed")
|
||||
else:
|
||||
patched = (
|
||||
src.replace("Path(resolved_module_file).resolve()", "Path(resolved_module_file)")
|
||||
.replace("Path(source_file).resolve()", "Path(source_file)")
|
||||
)
|
||||
assert patched != src, "FATAL: transformers symlink patch matched nothing"
|
||||
path.write_text(patched)
|
||||
print("patched transformers dynamic_module_utils.py (symlink hash fix)")
|
||||
PY
|
||||
|
||||
# -----------------------
|
||||
# Performance environment variable.
|
||||
|
||||
# Skip CuDNN compatibility check - not applicable for ROCm (uses MIOpen instead)
|
||||
ENV SGLANG_DISABLE_CUDNN_CHECK=1
|
||||
ENV HIP_FORCE_DEV_KERNARG=1
|
||||
ENV HSA_NO_SCRATCH_RECLAIM=1
|
||||
ENV SGLANG_ALLOW_OVERWRITE_LONGER_CONTEXT_LEN=1
|
||||
ENV SGLANG_INT4_WEIGHT=0
|
||||
ENV SGLANG_MOE_PADDING=1
|
||||
ENV SGLANG_ROCM_DISABLE_LINEARQUANT=0
|
||||
ENV SGLANG_ROCM_FUSED_DECODE_MLA=1
|
||||
ENV SGLANG_SET_CPU_AFFINITY=1
|
||||
ENV SGLANG_USE_AITER=1
|
||||
ENV SGLANG_USE_ROCM700A=1
|
||||
|
||||
ENV NCCL_MIN_NCHANNELS=112
|
||||
ENV ROCM_QUICK_REDUCE_QUANTIZATION=INT8
|
||||
ENV TORCHINDUCTOR_MAX_AUTOTUNE=1
|
||||
ENV TORCHINDUCTOR_MAX_AUTOTUNE_POINTWISE=1
|
||||
|
||||
CMD ["/bin/bash"]
|
||||
@@ -0,0 +1,6 @@
|
||||
FROM lmsysorg/sglang:latest
|
||||
|
||||
COPY serve /usr/bin/serve
|
||||
RUN chmod 777 /usr/bin/serve
|
||||
|
||||
ENTRYPOINT [ "/usr/bin/serve" ]
|
||||
Executable
+34
@@ -0,0 +1,34 @@
|
||||
#!/bin/bash
|
||||
echo "Starting server"
|
||||
|
||||
PREFIX="SM_SGLANG_"
|
||||
ARG_PREFIX="--"
|
||||
|
||||
ARGS=()
|
||||
|
||||
while IFS='=' read -r key value; do
|
||||
arg_name=$(echo "${key#"${PREFIX}"}" | tr '[:upper:]' '[:lower:]' | tr '_' '-')
|
||||
|
||||
ARGS+=("${ARG_PREFIX}${arg_name}")
|
||||
if [ -n "$value" ]; then
|
||||
ARGS+=("$value")
|
||||
fi
|
||||
done < <(env | grep "^${PREFIX}")
|
||||
|
||||
# Add default port only if not already set
|
||||
if ! [[ " ${ARGS[@]} " =~ " --port " ]]; then
|
||||
ARGS+=(--port "${SM_SGLANG_PORT:-8080}")
|
||||
fi
|
||||
|
||||
# Add default host only if not already set
|
||||
if ! [[ " ${ARGS[@]} " =~ " --host " ]]; then
|
||||
ARGS+=(--host "${SM_SGLANG_HOST:-0.0.0.0}")
|
||||
fi
|
||||
|
||||
# Add default model-path only if not already set
|
||||
if ! [[ " ${ARGS[@]} " =~ " --model-path " ]]; then
|
||||
ARGS+=(--model-path "${SM_SGLANG_MODEL_PATH:-/opt/ml/model}")
|
||||
fi
|
||||
|
||||
echo "Running command: exec python3 -m sglang.launch_server ${ARGS[@]}"
|
||||
exec python3 -m sglang.launch_server "${ARGS[@]}"
|
||||
@@ -0,0 +1,35 @@
|
||||
ARG BASE_IMG=pytorch/manylinux2_28-builder
|
||||
ARG CUDA_VERSION=13.0
|
||||
|
||||
FROM ${BASE_IMG}:cuda${CUDA_VERSION}
|
||||
|
||||
ARG ARCH=x86_64
|
||||
ARG CUDA_VERSION=13.0
|
||||
ARG PYTHON_VERSION=3.12
|
||||
ARG PYTHON_TAG=cp312-cp312
|
||||
ARG TORCH_VER=2.11.0
|
||||
ARG TVM_FFI_VER=0.1.11
|
||||
ARG PIP_DEFAULT_INDEX=https://pypi.python.org/simple
|
||||
ARG PYTORCH_MIRROR=download.pytorch.org
|
||||
|
||||
ENV PYTHON_ROOT_PATH=/opt/python/${PYTHON_TAG}
|
||||
ENV PATH=${PYTHON_ROOT_PATH}/bin:${PATH}
|
||||
|
||||
RUN yum install -y --nogpgcheck git wget tar gcc gcc-c++ make \
|
||||
&& yum clean all && rm -rf /var/cache/yum
|
||||
|
||||
RUN set -eux; \
|
||||
if [ "${ARCH}" = "aarch64" ]; then _LIB=sbsa; else _LIB="${ARCH}"; fi; \
|
||||
mkdir -p /usr/lib/${ARCH}-linux-gnu/; \
|
||||
ln -sf /usr/local/cuda-${CUDA_VERSION}/targets/${_LIB}-linux/lib/stubs/libcuda.so /usr/lib/${ARCH}-linux-gnu/libcuda.so
|
||||
|
||||
RUN --mount=type=cache,id=sgl-deep-gemm-pip,target=/root/.cache/pip \
|
||||
set -eux; \
|
||||
case "${CUDA_VERSION}" in \
|
||||
13.0) CU_TAG=cu130 ;; \
|
||||
12.9) CU_TAG=cu129 ;; \
|
||||
*) CU_TAG=cu130 ;; \
|
||||
esac; \
|
||||
${PYTHON_ROOT_PATH}/bin/pip install torch==${TORCH_VER} --index-url https://${PYTORCH_MIRROR}/whl/${CU_TAG}; \
|
||||
${PYTHON_ROOT_PATH}/bin/pip install --index-url ${PIP_DEFAULT_INDEX} \
|
||||
ninja setuptools wheel build numpy apache-tvm-ffi==${TVM_FFI_VER}
|
||||
@@ -0,0 +1,92 @@
|
||||
# Multi-stage build for sgl-router.
|
||||
#
|
||||
# Three stages, each scoped to its caching contract:
|
||||
# 1. chef — generate a `recipe.json` describing the dep graph.
|
||||
# 2. builder — compile deps from the recipe, then the workspace.
|
||||
# 3. runtime — distroless cc-debian12 with the stripped binary.
|
||||
#
|
||||
# The `cargo-chef` indirection is the canonical Rust multi-stage cache
|
||||
# pattern: the recipe step's inputs are JUST `Cargo.toml` + `Cargo.lock`,
|
||||
# so a source-only change produces a recipe-layer cache hit and the
|
||||
# heavy `cook --release` step is reused untouched. A naive "copy
|
||||
# manifests → cargo fetch → copy src" approach caches only the fetched
|
||||
# registry; every source change still recompiles every dep.
|
||||
#
|
||||
# `Cargo.lock` is gitignored repo-wide (root .gitignore "# Rust lib"
|
||||
# block), so we generate it inside the chef stage with `cargo
|
||||
# generate-lockfile` and propagate that lockfile to the builder via
|
||||
# `COPY --from=chef`. Both stages thus build against the same lockfile,
|
||||
# preserving --locked semantics within a single Docker build.
|
||||
#
|
||||
# Build (from the repo root):
|
||||
# docker build -f docker/sgl-router.Dockerfile -t sgl-router:dev .
|
||||
# Run:
|
||||
# docker run --rm -p 8090:8090 \
|
||||
# -v $(pwd)/docker/sgl-router.sample.yaml:/etc/sgl-router/sgl-router.yaml \
|
||||
# sgl-router:dev --config /etc/sgl-router/sgl-router.yaml
|
||||
#
|
||||
# Image budget: < 100 MB stripped (M6 acceptance). Verify with
|
||||
# `docker image inspect sgl-router:dev --format '{{.Size}}'`.
|
||||
|
||||
ARG RUST_VERSION=1.90
|
||||
ARG DEBIAN_VERSION=bookworm
|
||||
|
||||
######################## STAGE 1 — chef recipe ##########################
|
||||
FROM rust:${RUST_VERSION}-${DEBIAN_VERSION} AS chef
|
||||
RUN cargo install cargo-chef --locked --version ^0.1
|
||||
WORKDIR /work
|
||||
COPY experimental/sgl-router/Cargo.toml ./
|
||||
COPY experimental/sgl-router/rust-toolchain.toml ./
|
||||
# Stub a minimal src tree so cargo can resolve the workspace, generate
|
||||
# the lockfile (gitignored upstream), then prepare the chef recipe.
|
||||
RUN mkdir -p src && echo "fn main() {}" > src/main.rs \
|
||||
&& echo "" > src/lib.rs \
|
||||
&& cargo generate-lockfile \
|
||||
&& cargo chef prepare --recipe-path recipe.json \
|
||||
&& rm -rf src
|
||||
|
||||
######################## STAGE 2 — builder ##############################
|
||||
FROM rust:${RUST_VERSION}-${DEBIAN_VERSION} AS builder
|
||||
RUN cargo install cargo-chef --locked --version ^0.1
|
||||
WORKDIR /work
|
||||
|
||||
# `dynamo-tokenizers` pulls in `pcre2-sys`, whose build.rs links the SYSTEM
|
||||
# libpcre2-8 whenever pkg-config finds it (it does here — the rust:bookworm
|
||||
# base ships libpcre2-dev). That dynamic dep is absent from the distroless
|
||||
# runtime, so the binary fails at startup with "libpcre2-8.so.0: cannot open
|
||||
# shared object file". Force pcre2-sys to compile its vendored PCRE2 and link
|
||||
# it statically, keeping the runtime self-contained.
|
||||
ENV PCRE2_SYS_STATIC=1
|
||||
|
||||
COPY --from=chef /work/recipe.json ./recipe.json
|
||||
COPY --from=chef /work/Cargo.lock ./Cargo.lock
|
||||
COPY experimental/sgl-router/rust-toolchain.toml ./
|
||||
|
||||
# Cook (compile + cache) the dep graph from the recipe. This layer's
|
||||
# inputs are recipe.json + the toolchain — code changes in src/ do NOT
|
||||
# invalidate it.
|
||||
RUN cargo chef cook --release --recipe-path recipe.json
|
||||
|
||||
# Now bring in the real sources and the manifest they need.
|
||||
COPY experimental/sgl-router/Cargo.toml ./
|
||||
COPY experimental/sgl-router/src ./src
|
||||
|
||||
# --locked is intentionally omitted: the lockfile is generated in-container
|
||||
# (gitignored upstream) and `cargo chef cook` may have mutated it during the
|
||||
# dep-cook step, so a strict --locked check would spuriously fail.
|
||||
RUN cargo build --release --bin sgl-router \
|
||||
&& strip target/release/sgl-router
|
||||
|
||||
######################## STAGE 3 — runtime ##############################
|
||||
FROM gcr.io/distroless/cc-debian12:nonroot AS runtime
|
||||
|
||||
COPY --from=builder /work/target/release/sgl-router /usr/local/bin/sgl-router
|
||||
|
||||
# Default config path; mount your own via `-v <host-path>:/etc/sgl-router`.
|
||||
ENV SGL_ROUTER_CONFIG=/etc/sgl-router/sgl-router.yaml
|
||||
EXPOSE 8090
|
||||
|
||||
# distroless `nonroot` runs as uid 65532. The router doesn't need root.
|
||||
USER nonroot:nonroot
|
||||
|
||||
ENTRYPOINT ["/usr/local/bin/sgl-router"]
|
||||
@@ -0,0 +1,52 @@
|
||||
FROM ubuntu:24.04
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
ARG SGLANG_REPO=https://github.com/sgl-project/sglang.git
|
||||
ARG VER_SGLANG=main
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get full-upgrade -y && \
|
||||
DEBIAN_FRONTEND=noninteractive apt-get install --no-install-recommends -y \
|
||||
ca-certificates \
|
||||
git \
|
||||
curl \
|
||||
wget \
|
||||
vim \
|
||||
gcc \
|
||||
g++ \
|
||||
make \
|
||||
libsqlite3-dev \
|
||||
google-perftools \
|
||||
libtbb-dev \
|
||||
libnuma-dev \
|
||||
numactl
|
||||
|
||||
WORKDIR /opt
|
||||
|
||||
ENV UV_PYTHON_INSTALL_DIR=/usr/local/share/uv/python
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh && \
|
||||
mv /root/.local/bin/uv /root/.local/bin/uvx /usr/local/bin/ && \
|
||||
uv venv --python 3.12
|
||||
|
||||
RUN echo -e '[[index]]\nname = "torch"\nurl = "https://download.pytorch.org/whl/cpu"\n\n[[index]]\nname = "torchvision"\nurl = "https://download.pytorch.org/whl/cpu"\n\n[[index]]\nname = "torchaudio"\nurl = "https://download.pytorch.org/whl/cpu"\n\n[[index]]\nname = "triton"\nurl = "https://download.pytorch.org/whl/cpu"' > .venv/uv.toml
|
||||
|
||||
ENV UV_CONFIG_FILE=/opt/.venv/uv.toml
|
||||
|
||||
WORKDIR /sgl-workspace
|
||||
RUN source /opt/.venv/bin/activate && \
|
||||
git clone ${SGLANG_REPO} sglang && \
|
||||
cd sglang && \
|
||||
git checkout ${VER_SGLANG} && \
|
||||
cd python && \
|
||||
cp pyproject_cpu.toml pyproject.toml && \
|
||||
uv pip install . && \
|
||||
cd ../sgl-kernel && \
|
||||
cp pyproject_cpu.toml pyproject.toml && \
|
||||
uv pip install .
|
||||
|
||||
ENV SGLANG_USE_CPU_ENGINE=1
|
||||
ENV LD_PRELOAD=/usr/lib/x86_64-linux-gnu/libtcmalloc.so.4:/usr/lib/x86_64-linux-gnu/libtbbmalloc.so:/opt/.venv/lib/libiomp5.so
|
||||
ENV PATH="/opt/.venv/bin:$PATH"
|
||||
RUN echo 'source /opt/.venv/bin/activate' >> /root/.bashrc
|
||||
|
||||
WORKDIR /sgl-workspace/sglang
|
||||
@@ -0,0 +1,55 @@
|
||||
# docker build -t sglang:xpu -f xpu.Dockerfile --build-arg http_proxy=${http_proxy} --build-arg https_proxy=${https_proxy} --build-arg no_proxy=${no_proxy} --no-cache .
|
||||
|
||||
# Use Intel deep learning essentials base image with Ubuntu 24.04
|
||||
FROM intel/deep-learning-essentials:2025.3.2-0-devel-ubuntu24.04
|
||||
|
||||
# Avoid interactive prompts during package install
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# Define build arguments
|
||||
ARG PYTHON_VERSION=3.12
|
||||
|
||||
ARG SG_LANG_REPO=https://github.com/sgl-project/sglang.git
|
||||
ARG SG_LANG_BRANCH=main
|
||||
|
||||
ARG SG_LANG_KERNEL_REPO=https://github.com/sgl-project/sgl-kernel-xpu.git
|
||||
ARG SG_LANG_KERNEL_BRANCH=main
|
||||
|
||||
USER root
|
||||
|
||||
# Install the latest UMD driver for SYCL-TLA
|
||||
RUN apt-get update && apt-get install -y software-properties-common && \
|
||||
add-apt-repository -y ppa:kobuk-team/intel-graphics && \
|
||||
apt-get update && \
|
||||
apt-get install -y \
|
||||
libze-intel-gpu1 libze1 intel-metrics-discovery intel-opencl-icd clinfo intel-gsc \
|
||||
intel-media-va-driver-non-free libmfx-gen1 libvpl2 libvpl-tools libva-glx2 va-driver-all vainfo \
|
||||
libze-dev intel-ocloc && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
|
||||
RUN apt-get update && apt-get install -y \
|
||||
python3-dev \
|
||||
build-essential \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
ENV VIRTUAL_ENV="/opt/venv"
|
||||
ENV UV_PYTHON_INSTALL_DIR=/opt/uv/python
|
||||
RUN uv venv --python ${PYTHON_VERSION} --seed ${VIRTUAL_ENV}
|
||||
ENV PATH="$VIRTUAL_ENV/bin:$PATH"
|
||||
|
||||
WORKDIR /sgl-workspace
|
||||
|
||||
RUN pip install --no-cache-dir msgspec blake3 py-cpuinfo compressed_tensors gguf partial_json_parser einops tabulate --root-user-action=ignore && \
|
||||
pip install --no-cache-dir torch==2.12.0+xpu torchao==0.17.0+xpu torchvision==0.27.0+xpu torchaudio==2.11.0+xpu --index-url https://download.pytorch.org/whl/xpu
|
||||
|
||||
RUN echo "Cloning ${SG_LANG_BRANCH} from ${SG_LANG_REPO}" && \
|
||||
git clone --branch ${SG_LANG_BRANCH} --single-branch ${SG_LANG_REPO} sglang && \
|
||||
cd sglang && cd python && \
|
||||
cp pyproject_xpu.toml pyproject.toml && \
|
||||
pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu && \
|
||||
pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
||||
|
||||
CMD ["bash", "-c", "source /opt/intel/oneapi/setvars.sh --force && exec bash"]
|
||||
Reference in New Issue
Block a user