Sign inSign up

voipmonitor/sglang

By voipmonitor

•Updated 5 months ago

Image
0

10K+

voipmonitor/sglang repository overview

⁠voipmonitor/sglang:test-cu132

SGLang inference stack for RTX PRO 6000 Blackwell (SM120a) with NVFP4 support via b12x.

⁠Stack

  • Ubuntu 24.04 + CUDA 13.2.0 + cuDNN
  • PyTorch nightly (cu132)
  • FlashInfer (git main)
  • SGLang (source, editable, cherry-picks + patches)
  • b12x 0.5.1 (SM120-only NVFP4 MoE/GEMM)
  • nvidia-cutlass-dsl 4.4.2 (with CUDA 13.1 NVVM/ptxas fix)

⁠Dockerfile

# =============================================================================
# SGLang inference stack for RTX 5090 (Blackwell, sm_120a)
# torch 2.12.0 nightly (cu132) + sgl-kernel from source
# No vLLM — SGLang only, editable install (/opt/sglang)
# Base: Ubuntu 24.04 + CUDA 13.2.0 + cuDNN
# Stack: PyTorch nightly (cu132) -> Triton -> FlashInfer (git) ->
#        SGLang (source, editable, cherry-picks + patches)
#
# IMPORTANT — nvidia-cutlass-dsl packaging bug (as of 4.4.2):
#   The pip package nvidia-cutlass-dsl ships a native MLIR compiler blob
#   (_cutlass_ir.so) that statically links NVIDIA's proprietary NVVM compiler
#   and libnvptxcompiler (ptxas). These are closed-source — no public source
#   code exists, so the .so cannot be rebuilt from the CUTLASS git repo.
#
#   The package splits into two sub-packages:
#     - nvidia-cutlass-dsl-libs-base  (mandatory dep) — bundles NVVM/ptxas 12.9
#     - nvidia-cutlass-dsl-libs-cu13  (optional [cu13] extra) — bundles NVVM/ptxas 13.1
#
#   Both write the SAME file path. When installing "nvidia-cutlass-dsl[cu13]",
#   pip installs libs-base AFTER libs-cu13 (as its dependency), overwriting
#   the CUDA 13.1 .so with the CUDA 12.9 one.
#
#   ptxas 12.9 cannot compile _mma.block_scale instructions (private PTX)
#   that CUTLASS DSL generates for SM120 NVFP4 MoE/GEMM kernels (b12x).
#   ptxas 13.1 (from libs-cu13) can.
#
#   Fix: after every "pip install nvidia-cutlass-dsl[cu13]", force-reinstall
#   nvidia-cutlass-dsl-libs-cu13 as the LAST step so the 13.1 .so wins.
# =============================================================================

# -- args (override at build time) -------------------------------------------
ARG CUDA_VERSION=13.2.0
ARG UBUNTU_VERSION=24.04
ARG PYTHON_VERSION=3.12
ARG MAX_JOBS=128
ARG NVCC_THREADS=8
ARG TORCH_CUDA_ARCH_LIST="12.0a"
ARG FLASHINFER_CUDA_ARCH_LIST="12.0a"

# =============================================================================
# Stage 1: base – system packages, Python, uv
# =============================================================================
FROM nvidia/cuda:${CUDA_VERSION}-cudnn-devel-ubuntu${UBUNTU_VERSION} AS base

ARG PYTHON_VERSION
ARG DEBIAN_FRONTEND=noninteractive

ENV LANG=C.UTF-8 \
    LC_ALL=C.UTF-8 \
    PYTHONDONTWRITEBYTECODE=1 \
    PYTHONUNBUFFERED=1

# System dependencies
RUN apt-get update && apt-get install -y --no-install-recommends \
    python${PYTHON_VERSION} \
    python${PYTHON_VERSION}-dev \
    python${PYTHON_VERSION}-venv \
    python3-pip \
    git \
    git-lfs \
    wget \
    curl \
    build-essential \
    cmake \
    ninja-build \
    ccache \
    pkg-config \
    libssl-dev \
    libffi-dev \
    libjpeg-dev \
    libpng-dev \
    libgomp1 \
    numactl \
    libnuma-dev \
    libibverbs-dev \
    pciutils \
    && rm -rf /var/lib/apt/lists/*

# Make python3.12 the default
RUN update-alternatives --install /usr/bin/python3 python3 /usr/bin/python${PYTHON_VERSION} 1 && \
    update-alternatives --install /usr/bin/python python /usr/bin/python${PYTHON_VERSION} 1

# Install uv (fast Python package manager)
RUN curl -LsSf https://astral.sh/uv/install.sh | sh
ENV PATH="/root/.local/bin:${PATH}"

# Create venv with uv
RUN uv venv /opt/venv --python python${PYTHON_VERSION}
ENV VIRTUAL_ENV=/opt/venv \
    PATH="/opt/venv/bin:${PATH}"

# Ensure pip is available (needed for builds)
RUN uv pip install pip setuptools wheel

# ccache config for faster rebuilds
ENV CCACHE_DIR=/root/.ccache \
    CMAKE_C_COMPILER_LAUNCHER=ccache \
    CMAKE_CXX_COMPILER_LAUNCHER=ccache \
    CMAKE_CUDA_COMPILER_LAUNCHER=ccache

# =============================================================================
# Stage 2: deps – PyTorch stable + Triton + base packages
# =============================================================================
FROM base AS deps

ARG TORCH_CUDA_ARCH_LIST
ARG FLASHINFER_CUDA_ARCH_LIST
ARG MAX_JOBS
ARG NVCC_THREADS

ENV TORCH_CUDA_ARCH_LIST=${TORCH_CUDA_ARCH_LIST} \
    FLASHINFER_CUDA_ARCH_LIST=${FLASHINFER_CUDA_ARCH_LIST} \
    MAX_JOBS=${MAX_JOBS} \
    NVCC_THREADS=${NVCC_THREADS}

# CACHEBUST forces fresh clones every build
ARG CACHEBUST=0
WORKDIR /build

# Clone CUTLASS from git main (latest headers for FlashInfer)
RUN git clone --depth 1 https://github.com/NVIDIA/cutlass.git /build/cutlass && \
    grep "CUTLASS_MAJOR\|CUTLASS_MINOR\|CUTLASS_PATCH" /build/cutlass/include/cutlass/version.h | head -3

# -- Install PyTorch (nightly 2.12.0, cu132) ---------------------------------
RUN uv pip install \
    torch \
    --index-url https://download.pytorch.org/whl/nightly/cu132 \
    --force-reinstall
# torchvision needed by SGLang multimodal; not yet built for cu132,
# install from cu132 index (ships cu130 builds) without pulling deps
RUN uv pip install torchvision \
    --index-url https://download.pytorch.org/whl/nightly/cu132 \
    --prerelease=allow --no-deps

# -- Install latest NCCL for CUDA 13 -----------------------------------------
RUN uv pip install --force-reinstall nvidia-nccl-cu13

# Verify torch is installed
RUN python -c "import torch; v=torch.__version__; print(f'PyTorch: {v}')"

# -- Install Triton (from PyTorch nightly index) ------------------------------
RUN uv pip install triton --index-url https://download.pytorch.org/whl/nightly/cu132 --force-reinstall

# -- Install CUTLASS 4.x DSL (latest) ----------------------------------------
# NOTE: libs-base overwrites _cutlass_ir.so with NVVM 12.9; reinstall libs-cu13
# last so the CUDA 13.1 NVVM/ptxas binary wins (needed for _mma.block_scale on SM120)
RUN uv pip install "nvidia-cutlass-dsl[cu13]" && \
    pip install --force-reinstall --no-deps nvidia-cutlass-dsl-libs-cu13

# -- Install transformers (stable from PyPI) ----------------------------------
RUN uv pip install transformers

# -- Install other bleeding-edge packages ------------------------------------
RUN uv pip install --no-deps --upgrade \
    accelerate \
    safetensors \
    tokenizers \
    sentencepiece \
    bitsandbytes

# xformers (nightly, matching torch 2.12)
RUN uv pip install xformers --index-url https://download.pytorch.org/whl/nightly/cu132 --no-deps || true

# fastsafetensors
RUN uv pip install fastsafetensors

# Additional packages
RUN uv pip install xgrammar setproctitle gguf dill

# Upgrade cmake (sgl-kernel needs cmake >= 3.26 with CMP0169 support)
RUN uv pip install --upgrade cmake

# Verify torch is still in place
RUN python -c "import torch; v=torch.__version__; print(f'PyTorch {v}'); assert 'cu132' in v, f'WRONG TORCH: {v}'"

# =============================================================================
# Stage 3: build FlashInfer from source (JIT mode, git main)
# =============================================================================
FROM deps AS flashinfer-build

WORKDIR /build
RUN git clone --recursive https://github.com/flashinfer-ai/flashinfer.git

WORKDIR /build/flashinfer
# Cherry-pick unmerged PRs:
#   PR #2780 - fix(jit): enable GDC for CUTLASS GEMM PDL — SM100 flag only
RUN git config user.email "build@docker" && git config user.name "build" && \
    git fetch origin pull/2780/head:pr-2780 && \
    git cherry-pick --no-commit 8e78583b 7fcc7a7c

# Install FlashInfer's runtime deps
# Upgrade setuptools — FlashInfer's pyproject.toml uses PEP 639 license format
RUN uv pip install --upgrade setuptools pip && \
    uv pip install "apache-tvm-ffi>=0.1.6" "nvidia-cudnn-frontend>=1.13.0" packaging
RUN pip install -v --no-build-isolation --no-deps .

# =============================================================================
# Stage 4: SGLang from source (sgl-kernel + editable install)
# =============================================================================
FROM flashinfer-build AS sglang-build

ARG MAX_JOBS
ARG NVCC_THREADS

RUN git clone --recursive https://github.com/sgl-project/sglang.git /opt/sglang

# Cherry-pick unmerged PRs:
#   PR #19948 - Auto-disable DEEPGEMM_SCALE_UE8M0 for non-ue8m0 checkpoints on Blackwell
#   PR #19897 - Fix EAGLE-v2 NaN on radix cache prefix hits (zero-fill draft KV)
#   PR #19963 - Fix missing arch suffix in _get_cuda_arch_list for Blackwell/Hopper JIT
#   PR #20439 - Fix AWQ quantization ignoring modules_to_not_convert for FusedMoE layers
#   PR #20059 - Fuse normal_decode_set_metadata into Triton kernel (~2x decode metadata speedup)
#   PR #20074 - Qwen3.5: Fuse DeltaNet input projections (9% decode speedup)
#   PR #20110 - Qwen3.5: Fix MoE double allreduce with --enable-flashinfer-allreduce-fusion
#   PR #20114 - Fix: support HybridLinearAttnBackend in TboAttnBackend
#   PR #20182 - Fix mamba memory leak
#   PR #20232 - Fix Qwen3.5 fuse_moe_triton_tune bug
#   PR #20275 - Add GLM-5 thinking budget logit processor
#   PR #20377 - Remove mamba slot alloc in prefix match stage
#   PR #20433 - Mamba: make extra buffer track indexes async (15.7% throughput gain)
#   PR #20441 - Fix Piecewise CUDA Graph crash with --enable-mixed-chunk
#   PR #20445 - Avoid GPU syncs in mamba track metadata
#   PR #20448 - Qwen3.5 bugfix: Add mm_input_embeds in piecewise CUDA graph replay
#   PR #20460 - HiCache: Add synchronization for context parallelism
#   PR #20479 - Support Triton MLA FP8 KV cache (SM120)
#   PR #20534 - Transfer FP8 K/K_scale for CP indexer prefill gather
#   PR #20949 - Fix: warm up cuBLAS/cuBLASLt handles for FP16/BF16 (CUDA graph lazy-init fix)
# Merged (removed): #19977, #20103, #20266, #20390, #20396, #20539, #20604, #19961
WORKDIR /opt/sglang
RUN git config user.email "build@docker" && git config user.name "build" && \
    git fetch origin \
        pull/19897/head:pr-19897 \
        pull/19948/head:pr-19948 \
        pull/19963/head:pr-19963 \
        pull/20439/head:pr-20439 \
        pull/20059/head:pr-20059 \
        pull/20074/head:pr-20074 \
        pull/20110/head:pr-20110 \
        pull/20114/head:pr-20114 \
        pull/20182/head:pr-20182 \
        pull/20232/head:pr-20232 \
        pull/20275/head:pr-20275 \
        pull/20377/head:pr-20377 \
        pull/20433/head:pr-20433 \
        pull/20441/head:pr-20441 \
        pull/20445/head:pr-20445 \
        pull/20448/head:pr-20448 \
        pull/20460/head:pr-20460 \
        pull/20479/head:pr-20479 \
        pull/20534/head:pr-20534 \
        pull/20949/head:pr-20949 && \
    (git cherry-pick --no-commit 97d666ae5109 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit fc05fc6758aa || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 340f619 b38295b 0ba073f || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 23aa58fa1b7c || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 1e5d20c64aac e0137cc66274 3345601bab22 556f00b70aec 39f312d94c5e || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 4972ab6e76bf b77762f28fb6 3da85d453dd2 d2d3f022be27 5bc1fac6230f 6dcc77825b10 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 140be31a7951 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit b961a4124855 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit a38e09ac8ace 6be4ccd441fb 27d36e52e1ee || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 6c6f2c8e68f6 29066eb35817 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 45baf2e56ab5 42b872677423 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit e7258d164d1b 746f28699356 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 0433bc88b5ae dc0a80235e91 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit ace5f94982a9 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit cc4a43a1698b || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 6bcb570f95a8 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 614b7815b53f || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 3019db47e505 64cc6ffa1d4d 3e8774f6f6d1 0e25a1cb74c4 b6551f2ba4af c69aa5debd47 75d4edbbadf9 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit 4578257732bf f695b901400833 || git cherry-pick --abort 2>/dev/null || true) && \
    (git cherry-pick --no-commit f6c25c12c || git cherry-pick --abort 2>/dev/null || true)

# Cherry-pick PCIe allreduce + b12x integration (lukealonso fork)
# NOTE: v2 commit d39236aee635 has merge conflicts with current main, using v1 until rebase
# Order: PCIe cherry-pick → resolve conflicts → b12x cherry-pick → resolve conflicts
COPY patches/fix-pcie-allreduce-conflict.py /tmp/fix-cherry-pick-conflicts.py
RUN git remote add lukealonso https://github.com/lukealonso/sglang.git && \
    git fetch lukealonso 5bb89b03afe46fbd012da9f50bb5992673342123 c70c22a15a16961c65b8c61f7f18fae07995ec38 && \
    (git cherry-pick --no-commit 5bb89b03afe46fbd012da9f50bb5992673342123 || git cherry-pick --abort 2>/dev/null || true) && \
    python /tmp/fix-cherry-pick-conflicts.py && \
    git add -A && \
    (git cherry-pick --no-commit c70c22a15a16961c65b8c61f7f18fae07995ec38 || git cherry-pick --abort 2>/dev/null || true) && \
    python /tmp/fix-cherry-pick-conflicts.py && \
    git add -A && \
    rm /tmp/fix-cherry-pick-conflicts.py

# -- Patch sgl-kernel: remove sm_100a/sm_103a (only need sm_120a + sm_90a) ----
RUN sed -i '/compute_100a.*sm_100a/d' sgl-kernel/CMakeLists.txt && \
    sed -i '/compute_103a.*sm_103a/d' sgl-kernel/CMakeLists.txt && \
    sed -i '/compute_100a/d' sgl-kernel/cmake/flashmla.cmake && \
    sed -i '/compute_103a/d' sgl-kernel/cmake/flashmla.cmake && \
    echo "OK: removed sm_100a/sm_103a gencode from sgl-kernel"

# Build sgl-kernel for Blackwell (sm_120a) + Hopper FA3 (sm_90a)
WORKDIR /opt/sglang/sgl-kernel
RUN uv pip install scikit-build-core
RUN --mount=type=cache,target=/root/.ccache \
    TORCH_CUDA_ARCH_LIST="12.0a" \
    MAX_JOBS=${MAX_JOBS} \
    make build CMAKE_ARGS="-DCMAKE_C_COMPILER_LAUNCHER=ccache -DCMAKE_CXX_COMPILER_LAUNCHER=ccache -DENABLE_BELOW_SM90=0"

# Install SGLang Python package (editable so source tree is live)
# --no-deps: prevents pip from pulling stable torch, sgl-kernel, flashinfer from PyPI
WORKDIR /opt/sglang
RUN pip install --no-build-isolation --no-deps -e "./python[all]"

# Install SGLang's runtime deps (explicit list, excludes source-built packages)
RUN uv pip install \
        ipython pynvml orjson "apache-tvm-ffi>=0.1.6" \
        openai anthropic aiohttp fastapi uvicorn uvloop \
        numpy scipy pandas pillow requests psutil pydantic \
        packaging tqdm regex interegular outlines compressed-tensors \
        lark-parser cloudpickle rpyc msgspec \
        multipart python-multipart decord soundfile librosa \
        prometheus-client prometheus-fastapi-instrumentator \
        partial_json_parser tiktoken modelscope pybase64 \
        huggingface_hub \
        xgrammar setproctitle gguf dill pyzmq einops openai-harmony \
    || true

# Verify torch wasn't overwritten
RUN python -c "import torch; v=torch.__version__; print(f'PyTorch {v}'); assert 'cu132' in v, f'TORCH DOWNGRADED: {v}'"

# Reinstall bleeding-edge packages (--no-deps to be safe)
RUN uv pip install --upgrade huggingface_hub && \
    uv pip install --upgrade transformers && \
    uv pip install --force-reinstall "nvidia-cutlass-dsl[cu13]"

# Fix nvidia-cutlass-dsl: libs-base ships _cutlass_ir.so with NVVM/ptxas 12.9
# which cannot compile _mma.block_scale instructions for SM120.
# libs-cu13 has the correct NVVM/ptxas 13.1 but libs-base overwrites it.
# Reinstalling libs-cu13 last ensures the CUDA 13 .so wins.
RUN pip install --force-reinstall --no-deps nvidia-cutlass-dsl-libs-cu13 && \
    python -c "from cutlass._mlir._mlir_libs import _cutlass_ir; print('OK: cutlass MLIR loaded')"

# Patch cutlass-dsl: copy blockscaled_layout.py from git main (adds sm120_make_smem_layout_sfa)
RUN CUTLASS_DSL=$(python -c 'import cutlass.utils.blockscaled_layout as m; print(m.__file__)') && \
    cp /build/cutlass/python/CuTeDSL/cutlass/utils/blockscaled_layout.py "$CUTLASS_DSL" && \
    python -c 'from cutlass.utils.blockscaled_layout import sm120_make_smem_layout_sfa; print("OK: sm120_make_smem_layout_sfa available")'

# b12x: TP-only NVFP4 MoE/GEMM backend (no-deps to avoid torch conflicts)
RUN uv pip install b12x --no-deps

# Verify SGLang
RUN python -c "import sglang; print(f'SGLang {sglang.__version__}')"

# Copy Triton MoE configs for Blackwell GPUs
COPY configs/ /tmp/triton-configs/
RUN SGLANG_PKG="/opt/sglang/python" && \
    TRITON_CONFIGS_DIR="${SGLANG_PKG}/sglang/srt/layers/moe/fused_moe_triton/configs/triton_3_6_0" && \
    mkdir -p "$TRITON_CONFIGS_DIR" && \
    for f in /tmp/triton-configs/*.json; do \
        base=$(basename "$f"); \
        cp "$f" "$TRITON_CONFIGS_DIR/$base"; \
        cp "$f" "${TRITON_CONFIGS_DIR}/$(echo "$base" | sed 's/\.json$/_down.json/')"; \
        nodt=$(echo "$base" | sed 's/,dtype=[^.]*//'); \
        cp "$f" "$TRITON_CONFIGS_DIR/$nodt"; \
        cp "$f" "${TRITON_CONFIGS_DIR}/$(echo "$nodt" | sed 's/\.json$/_down.json/')"; \
    done && \
    rm -rf /tmp/triton-configs && \
    ls -la "$TRITON_CONFIGS_DIR/"

# -- Upgrade FlashInfer CUTLASS headers to latest (git main) -------------------
RUN FLASHINFER_CUTLASS="$(python -c "import flashinfer, os; \
    base = os.path.dirname(flashinfer.__file__); \
    candidates = [os.path.join(base, s) for s in ['data/cutlass', 'cutlass']]; \
    print(next(p for p in candidates if os.path.isdir(p)))")" && \
    echo "CUTLASS dir: ${FLASHINFER_CUTLASS}" && \
    rm -rf "${FLASHINFER_CUTLASS}/include" "${FLASHINFER_CUTLASS}/tools" && \
    cp -a /build/cutlass/include "${FLASHINFER_CUTLASS}/include" && \
    cp -a /build/cutlass/tools "${FLASHINFER_CUTLASS}/tools" && \
    grep "CUTLASS_MAJOR\|CUTLASS_MINOR\|CUTLASS_PATCH" "${FLASHINFER_CUTLASS}/include/cutlass/version.h" | head -3

# -- Upgrade cuDNN to latest ------------------------------------------------
RUN uv pip install --upgrade nvidia-cudnn-cu13 && \
    python -c "import torch; print(f'cuDNN: {torch.backends.cudnn.version()}')"

# Remove CUDA 12 pip packages
RUN pip uninstall -y \
    nvidia-cublas-cu12 nvidia-cuda-cupti-cu12 nvidia-cuda-nvrtc-cu12 \
    nvidia-cuda-runtime-cu12 nvidia-cudnn-cu12 nvidia-cufft-cu12 \
    nvidia-cufile-cu12 nvidia-curand-cu12 nvidia-cusolver-cu12 \
    nvidia-cusparse-cu12 nvidia-cusparselt-cu12 nvidia-nccl-cu12 \
    nvidia-nvjitlink-cu12 nvidia-nvshmem-cu12 nvidia-nvtx-cu12 \
    2>/dev/null || true && \
    pip install --force-reinstall \
    nvidia-nccl-cu13 nvidia-cusparselt-cu13 nvidia-nvshmem-cu13 nvidia-cudnn-cu13 \
    2>/dev/null || true && \
    python -c "import torch; print(f'torch {torch.__version__} OK')" && \
    echo "Verifying no CUDA 12 runtime:" && \
    ! find /opt/venv -name 'libcudart.so.12*' 2>/dev/null | grep -q . && \
    echo "OK: no libcudart.so.12 found"

# Clean up pip/uv caches before copying to final stage
RUN pip cache purge 2>/dev/null || true && \
    uv cache clean 2>/dev/null || true && \
    rm -rf /root/.cache/pip /root/.cache/uv /tmp/pip-* && \
    find /opt/venv -type d -name "__pycache__" -exec rm -rf {} + 2>/dev/null || true && \
    find /opt/venv -name "*.pyc" -delete 2>/dev/null || true && \
    find /opt/venv -name "*.a" -delete 2>/dev/null || true

# =============================================================================
# Stage 5: final – runtime image with CUDA devel (needed for FlashInfer JIT)
# =============================================================================
FROM nvidia/cuda:${CUDA_VERSION}-cudnn-devel-ubuntu${UBUNTU_VERSION} AS final

ARG PYTHON_VERSION=3.12
ARG TORCH_CUDA_ARCH_LIST
ARG FLASHINFER_CUDA_ARCH_LIST
ARG DEBIAN_FRONTEND=noninteractive

# Runtime + JIT compilation deps (FlashInfer JIT needs nvcc, gcc, Python headers)
RUN apt-get update && apt-get install -y --no-install-recommends \
    python${PYTHON_VERSION} \
    python${PYTHON_VERSION}-venv \
    python${PYTHON_VERSION}-dev \
    gcc \
    g++ \
    libc6-dev \
    libgomp1 \
    numactl \
    libnuma1 \
    libibverbs1 \
    libjpeg8 \
    libpng16-16t64 \
    ninja-build \
    pciutils \
    curl \
    && rm -rf /var/lib/apt/lists/*

RUN update-alternatives --install /usr/bin/python3 python3 /usr/bin/python${PYTHON_VERSION} 1 && \
    update-alternatives --install /usr/bin/python python /usr/bin/python${PYTHON_VERSION} 1

# Copy venv and SGLang source (editable install needs the source tree)
COPY --from=sglang-build /opt/venv /opt/venv
COPY --from=sglang-build /opt/sglang /opt/sglang

ENV VIRTUAL_ENV=/opt/venv \
    PATH="/opt/venv/bin:${PATH}" \
    LANG=C.UTF-8 \
    LC_ALL=C.UTF-8 \
    PYTHONDONTWRITEBYTECODE=1 \
    PYTHONUNBUFFERED=1

# Runtime env
ENV TORCH_CUDA_ARCH_LIST=${TORCH_CUDA_ARCH_LIST} \
    FLASHINFER_CUDA_ARCH_LIST=${FLASHINFER_CUDA_ARCH_LIST} \
    NCCL_P2P_DISABLE=0 \
    NCCL_IB_DISABLE=0 \
    CUDA_DEVICE_MAX_CONNECTIONS=32

# JIT cache directories
ARG BUILD_ID=unknown
RUN echo "${BUILD_ID}" > /etc/jit-build-id
ENV JIT_BUILD_ID=${BUILD_ID} \
    TRITON_CACHE_DIR=/cache/jit/triton \
    TORCH_EXTENSIONS_DIR=/cache/jit/torch_extensions \
    FLASHINFER_WORKSPACE_BASE=/cache/jit/flashinfer \
    TVM_FFI_CACHE_DIR=/cache/jit/tvm-ffi \
    XDG_CACHE_HOME=/cache/jit \
    HF_HOME=/root/.cache/huggingface

RUN mkdir -p /cache/jit/triton /cache/jit/torch_extensions

# Ensure pip nvidia libs load before system CUDA (pip cuBLAS may be newer)
ENV LD_LIBRARY_PATH="/opt/venv/lib/python3.12/site-packages/nvidia/cu13/lib:/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"

# JIT cache invalidation
COPY jit-cache-invalidate.sh /usr/local/bin/jit-cache-invalidate.sh
RUN chmod +x /usr/local/bin/jit-cache-invalidate.sh && \
    echo 'source /usr/local/bin/jit-cache-invalidate.sh' >> /root/.bashrc
ENV BASH_ENV=/usr/local/bin/jit-cache-invalidate.sh

COPY entrypoint-sglang.sh /entrypoint.sh
COPY scripts/sglang_kld_eval.py /workspace/sglang_kld_eval.py

WORKDIR /workspace

# -- Runtime patches (applied after build, before smoke test) -----------------
COPY patches/sglang-kld-logit-capture.py /tmp/patches/
COPY patches/fix-glm-moe-strip-whitespace.py /tmp/patches/
COPY patches/apply-indexcache.py /tmp/patches/
COPY patches/fix-blackwell-flashinfer-backend-assert.py /tmp/patches/
COPY patches/fix-pcie-allreduce-conflict.py /tmp/patches/
RUN python /tmp/patches/sglang-kld-logit-capture.py && \
    python /tmp/patches/fix-glm-moe-strip-whitespace.py && \
    python /tmp/patches/apply-indexcache.py && \
    python /tmp/patches/fix-blackwell-flashinfer-backend-assert.py && \
    python /tmp/patches/fix-pcie-allreduce-conflict.py && \
    rm -rf /tmp/patches

# Smoke test (import only, no GPU needed)
RUN python -c "\
import torch; \
import flashinfer; \
import transformers; \
import sglang; \
print(f'PyTorch:      {torch.__version__}'); \
print(f'CUDA:         {torch.version.cuda}'); \
print(f'FlashInfer:   {flashinfer.__version__}'); \
print(f'Transformers: {transformers.__version__}'); \
print(f'SGLang:       {getattr(sglang, \"__version__\", \"editable-install\")}'); \
print(f'DeepGEMM:     bundled in sgl-kernel'); \
assert 'cu132' in torch.__version__, f'WRONG TORCH IN FINAL IMAGE: {torch.__version__}'; \
print('OK: torch nightly cu132, sgl-kernel from source, no vLLM'); \
"

ENTRYPOINT ["/entrypoint.sh"]

Tag summary

Content type

Image

Digest

sha256:6913b65fa…

Size

8.5 GB

Last updated

5 months ago

docker pull voipmonitor/sglang:glm51-luke-sync-a16off-20260511