text-generation-inference

text-generation-inference Dockerfile

7.5 KB 231 lines Raw ↗ GitHub ↗

# Rust builderFROM lukemathwalker/cargo-chef:latest-rust-1.71 AS chefWORKDIR /usr/srcARG CARGO_REGISTRIES_CRATES_IO_PROTOCOL=sparseFROM chef as plannerCOPY Cargo.toml Cargo.tomlCOPY rust-toolchain.toml rust-toolchain.tomlCOPY proto protoCOPY benchmark benchmarkCOPY router routerCOPY launcher launcherRUN cargo chef prepare --recipe-path recipe.jsonFROM chef AS builderARG GIT_SHAARG DOCKER_LABELRUN PROTOC_ZIP=protoc-21.12-linux-x86_64.zip && \    curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v21.12/$PROTOC_ZIP && \    unzip -o $PROTOC_ZIP -d /usr/local bin/protoc && \    unzip -o $PROTOC_ZIP -d /usr/local 'include/*' && \    rm -f $PROTOC_ZIPCOPY --from=planner /usr/src/recipe.json recipe.jsonRUN cargo chef cook --release --recipe-path recipe.jsonCOPY Cargo.toml Cargo.tomlCOPY rust-toolchain.toml rust-toolchain.tomlCOPY proto protoCOPY benchmark benchmarkCOPY router routerCOPY launcher launcherRUN cargo build --release# Python builder# Adapted from: https://github.com/pytorch/pytorch/blob/master/DockerfileFROM debian:bullseye-slim as pytorch-installARG PYTORCH_VERSION=2.0.1ARG PYTHON_VERSION=3.9# Keep in sync with `server/pyproject.tomlARG CUDA_VERSION=11.8ARG MAMBA_VERSION=23.1.0-1ARG CUDA_CHANNEL=nvidiaARG INSTALL_CHANNEL=pytorch# Automatically set by buildxARG TARGETPLATFORMENV PATH /opt/conda/bin:$PATHRUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \        build-essential \        ca-certificates \        ccache \        curl \        git && \        rm -rf /var/lib/apt/lists/*# Install conda# translating Docker's TARGETPLATFORM into mamba archesRUN case ${TARGETPLATFORM} in \         "linux/arm64")  MAMBA_ARCH=aarch64  ;; \         *)              MAMBA_ARCH=x86_64   ;; \    esac && \    curl -fsSL -v -o ~/mambaforge.sh -O  "https://github.com/conda-forge/miniforge/releases/download/${MAMBA_VERSION}/Mambaforge-${MAMBA_VERSION}-Linux-${MAMBA_ARCH}.sh"RUN chmod +x ~/mambaforge.sh && \    bash ~/mambaforge.sh -b -p /opt/conda && \    rm ~/mambaforge.sh# Install pytorch# On arm64 we exit with an error codeRUN case ${TARGETPLATFORM} in \         "linux/arm64")  exit 1 ;; \         *)              /opt/conda/bin/conda update -y conda &&  \                         /opt/conda/bin/conda install -c "${INSTALL_CHANNEL}" -c "${CUDA_CHANNEL}" -y "python=${PYTHON_VERSION}" pytorch==$PYTORCH_VERSION "pytorch-cuda=$(echo $CUDA_VERSION | cut -d'.' -f 1-2)"  ;; \    esac && \    /opt/conda/bin/conda clean -ya# CUDA kernels builder imageFROM pytorch-install as kernel-builderRUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \        ninja-build \        && rm -rf /var/lib/apt/lists/*RUN /opt/conda/bin/conda install -c "nvidia/label/cuda-11.8.0"  cuda==11.8 && \    /opt/conda/bin/conda clean -ya# Build Flash Attention CUDA kernelsFROM kernel-builder as flash-att-builderWORKDIR /usr/srcCOPY server/Makefile-flash-att Makefile# Build specific version of flash attentionRUN make build-flash-attention# Build Flash Attention v2 CUDA kernelsFROM kernel-builder as flash-att-v2-builderWORKDIR /usr/srcCOPY server/Makefile-flash-att-v2 Makefile# Build specific version of flash attention v2RUN make build-flash-attention-v2# Build Transformers exllama kernelsFROM kernel-builder as exllama-kernels-builderWORKDIR /usr/srcCOPY server/exllama_kernels/ .# Build specific version of transformersRUN TORCH_CUDA_ARCH_LIST="8.0;8.6+PTX" python setup.py build# Build Transformers awq kernelsFROM kernel-builder as awq-kernels-builderWORKDIR /usr/srcCOPY server/Makefile-awq Makefile# Build specific version of transformersRUN TORCH_CUDA_ARCH_LIST="8.0;8.6+PTX" make build-awq# Build eetq kernelsFROM kernel-builder as eetq-kernels-builderWORKDIR /usr/srcCOPY server/Makefile-eetq Makefile# Build specific version of transformersRUN TORCH_CUDA_ARCH_LIST="8.0;8.6+PTX" make build-eetq# Build Transformers CUDA kernelsFROM kernel-builder as custom-kernels-builderWORKDIR /usr/srcCOPY server/custom_kernels/ .# Build specific version of transformersRUN python setup.py build# Build vllm CUDA kernelsFROM kernel-builder as vllm-builderWORKDIR /usr/srcCOPY server/Makefile-vllm Makefile# Build specific version of vllmRUN make build-vllm# Text Generation Inference base imageFROM nvidia/cuda:11.8.0-base-ubuntu20.04 as base# Conda envENV PATH=/opt/conda/bin:$PATH \    CONDA_PREFIX=/opt/conda# Text Generation Inference base envENV HUGGINGFACE_HUB_CACHE=/data \    HF_HUB_ENABLE_HF_TRANSFER=1 \    PORT=80WORKDIR /usr/srcRUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \        libssl-dev \        ca-certificates \        make \        curl \        && rm -rf /var/lib/apt/lists/*# Copy conda with PyTorch installedCOPY --from=pytorch-install /opt/conda /opt/conda# Copy build artifacts from flash attention builderCOPY --from=flash-att-builder /usr/src/flash-attention/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packagesCOPY --from=flash-att-builder /usr/src/flash-attention/csrc/layer_norm/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packagesCOPY --from=flash-att-builder /usr/src/flash-attention/csrc/rotary/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packages# Copy build artifacts from flash attention v2 builderCOPY --from=flash-att-v2-builder /usr/src/flash-attention-v2/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packages# Copy build artifacts from custom kernels builderCOPY --from=custom-kernels-builder /usr/src/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packages# Copy build artifacts from exllama kernels builderCOPY --from=exllama-kernels-builder /usr/src/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packages# Copy build artifacts from awq kernels builderCOPY --from=awq-kernels-builder /usr/src/llm-awq/awq/kernels/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packages# Copy build artifacts from eetq kernels builderCOPY --from=eetq-kernels-builder /usr/src/eetq/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packages# Copy builds artifacts from vllm builderCOPY --from=vllm-builder /usr/src/vllm/build/lib.linux-x86_64-cpython-39 /opt/conda/lib/python3.9/site-packages# Install flash-attention dependenciesRUN pip install einops --no-cache-dir# Install serverCOPY proto protoCOPY server serverCOPY server/Makefile server/MakefileRUN cd server && \    make gen-server && \    pip install -r requirements.txt && \    pip install ".[bnb, accelerate, quantize]" --no-cache-dir# Install benchmarkerCOPY --from=builder /usr/src/target/release/text-generation-benchmark /usr/local/bin/text-generation-benchmark# Install routerCOPY --from=builder /usr/src/target/release/text-generation-router /usr/local/bin/text-generation-router# Install launcherCOPY --from=builder /usr/src/target/release/text-generation-launcher /usr/local/bin/text-generation-launcherRUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \        build-essential \        g++ \        && rm -rf /var/lib/apt/lists/*# AWS Sagemaker compatbile imageFROM base as sagemakerCOPY sagemaker-entrypoint.sh entrypoint.shRUN chmod +x entrypoint.shENTRYPOINT ["./entrypoint.sh"]# Final imageFROM baseENTRYPOINT ["text-generation-launcher"]CMD ["--json-output"]