mirror of
https://github.com/modelscope/modelscope.git
synced 2026-09-01 19:49:03 +02:00
merge master
This commit is contained in:
26
docker/Dockerfile.amd
Normal file
26
docker/Dockerfile.amd
Normal file
@@ -0,0 +1,26 @@
|
||||
FROM {base_image}
|
||||
|
||||
ARG BASE_IMAGE_TAG={base_image_tag}
|
||||
LABEL modelscope.base_image="vllm/vllm-openai-rocm:${BASE_IMAGE_TAG}"
|
||||
|
||||
# Build-time only (ARG does not persist into the image). Aliyun mirror can lag PyPI (~1h).
|
||||
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
|
||||
|
||||
COPY docker/scripts/modelscope_env_init.sh /usr/local/bin/ms_env_init.sh
|
||||
|
||||
ARG CUR_TIME={cur_time}
|
||||
RUN echo "CUR_TIME=${CUR_TIME}" && echo "BASE_IMAGE_TAG=${BASE_IMAGE_TAG}"
|
||||
|
||||
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
pip config set global.index-url https://mirrors.aliyun.com/pypi/simple && \
|
||||
pip config set install.trusted-host mirrors.aliyun.com && \
|
||||
cd /tmp && GIT_LFS_SKIP_SMUDGE=1 git clone -b {modelscope_branch} --single-branch https://github.com/modelscope/modelscope.git && \
|
||||
cd modelscope && pip install --no-cache-dir . -f https://modelscope.oss-cn-beijing.aliyuncs.com/releases/repo.html && \
|
||||
cd / && rm -fr /tmp/modelscope && pip cache purge
|
||||
|
||||
ENV VLLM_USE_MODELSCOPE=True
|
||||
ENV LMDEPLOY_USE_MODELSCOPE=True
|
||||
ENV MODELSCOPE_CACHE=/mnt/workspace/.cache/modelscope/hub
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
@@ -5,39 +5,54 @@ ENV PIP_DISABLE_PIP_VERSION_CHECK=1 \
|
||||
PIP_RETRIES=10 \
|
||||
SOC_VERSION={soc_version} \
|
||||
CANN_VERSION={cann_version}
|
||||
# Build-time only (ARG does not persist into the image).
|
||||
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
# ---------- System dependencies ----------
|
||||
RUN rm -f /etc/apt/apt.conf.d/docker-clean && \
|
||||
find /etc/apt/apt.conf.d -maxdepth 1 -type f | xargs -r grep -l "APT::Update::Post-Invoke\|docker-clean" | xargs -r rm -f && \
|
||||
apt-get update -y && \
|
||||
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
|
||||
gcc g++ cmake ninja-build libnuma-dev libgl1 libglib2.0-0 libsm6 libxext6 libxrender1 \
|
||||
wget git curl jq vim build-essential ca-certificates && \
|
||||
apt-get clean && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
RUN set -eux; \
|
||||
. /etc/os-release; \
|
||||
case "${ID,,}" in \
|
||||
ubuntu) \
|
||||
rm -f /etc/apt/apt.conf.d/docker-clean; \
|
||||
find /etc/apt/apt.conf.d -maxdepth 1 -type f | xargs -r grep -l "APT::Update::Post-Invoke\|docker-clean" | xargs -r rm -f; \
|
||||
apt-get update -y; \
|
||||
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
|
||||
gcc g++ cmake ninja-build libnuma-dev libgl1 libglib2.0-0 libsm6 libxext6 libxrender1 \
|
||||
wget git curl jq vim build-essential ca-certificates; \
|
||||
apt-get clean; \
|
||||
rm -rf /var/lib/apt/lists/* \
|
||||
;; \
|
||||
openeuler) \
|
||||
yum install -y \
|
||||
gcc gcc-c++ cmake ninja-build numactl-devel mesa-libGL glib2 libSM libXext libXrender \
|
||||
wget git curl jq vim make ca-certificates; \
|
||||
yum clean all; \
|
||||
rm -rf /var/cache/yum \
|
||||
;; \
|
||||
*) \
|
||||
echo "Unsupported base image OS: ${ID}" >&2; \
|
||||
exit 1 \
|
||||
;; \
|
||||
esac
|
||||
|
||||
RUN pip config set global.index-url https://mirrors.aliyun.com/pypi/simple && \
|
||||
pip config set global.extra-index-url "https://pypi.org/simple" && \
|
||||
pip config set install.trusted-host mirrors.aliyun.com && \
|
||||
ARCH=$(uname -m) && \
|
||||
if [ "$ARCH" = "x86_64" ]; then \
|
||||
pip config set global.extra-index-url "https://pypi.org/simple https://download.pytorch.org/whl/cpu/"; \
|
||||
fi
|
||||
pip config set install.trusted-host mirrors.aliyun.com
|
||||
|
||||
{extra_content}
|
||||
# ---------- Install vllm + vllm-ascend ----------
|
||||
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
|
||||
git clone --depth 1 --branch v0.18.0 https://github.com/vllm-project/vllm && \
|
||||
git clone --depth 1 --branch v0.18.0 https://github.com/vllm-project/vllm-ascend.git
|
||||
git clone --depth 1 --branch {vllm_git_ref} https://github.com/vllm-project/vllm && \
|
||||
git clone --depth 1 --branch {vllm_ascend_git_ref} https://github.com/vllm-project/vllm-ascend.git
|
||||
|
||||
RUN ARCH=$(uname -m) && \
|
||||
export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
source /usr/local/Ascend/nnal/atb/set_env.sh && \
|
||||
# Install torch & torch_npu & torchvision
|
||||
pip install torch==2.9.0 torch_npu==2.9.0.post2 torchvision==0.24.0 && \
|
||||
pip install torch=={torch_version} torch_npu=={torch_npu_version} torchvision=={torchvision_version} && \
|
||||
# Install vllm
|
||||
cd vllm && VLLM_TARGET_DEVICE=empty pip install -v -e . && cd .. && \
|
||||
# Install vllm-ascend
|
||||
@@ -46,32 +61,33 @@ RUN ARCH=$(uname -m) && \
|
||||
# ---------- Clone training-side repositories ----------
|
||||
RUN git clone --depth 1 --branch {megatron_branch} https://github.com/NVIDIA/Megatron-LM.git /Megatron-LM && \
|
||||
git clone --depth 1 --branch {mindspeed_branch} https://gitcode.com/Ascend/MindSpeed.git /MindSpeed && \
|
||||
GIT_LFS_SKIP_SMUDGE=1 git clone --depth 1 -b {swift_branch} --single-branch https://github.com/modelscope/ms-swift.git /ms-swift && \
|
||||
git clone --depth 1 https://github.com/modelscope/mcore-bridge.git /mcore-bridge
|
||||
GIT_LFS_SKIP_SMUDGE=1 git clone --depth 1 -b {swift_branch} --single-branch https://github.com/modelscope/ms-swift.git /ms-swift
|
||||
|
||||
# ---------- Install training-side repositories ----------
|
||||
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
|
||||
cd /MindSpeed && pip install --no-cache-dir -e . && \
|
||||
cd /mcore-bridge && pip install --no-cache-dir -e . && \
|
||||
pip install --no-cache-dir mcore-bridge -i https://pypi.org/simple/ -U && \
|
||||
cd /ms-swift && pip install --no-cache-dir -e .
|
||||
|
||||
# ---------- Pin torch to the correct version + torch_npu ----------
|
||||
# x86: must force-install the CPU build from pytorch.org/whl/cpu
|
||||
# aarch64: PyPI only provides the CPU build, so install it directly from the Aliyun mirror
|
||||
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
|
||||
ARCH=$(uname -m) && \
|
||||
if [ "$ARCH" = "x86_64" ]; then \
|
||||
pip install --no-cache-dir --force-reinstall --no-deps \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.9.0 torchvision==0.24.0 torchaudio==2.9.0; \
|
||||
torch=={torch_version} torchvision=={torchvision_version} torchaudio=={torchaudio_version}; \
|
||||
else \
|
||||
pip install --no-cache-dir --force-reinstall --no-deps \
|
||||
torch==2.9.0 torchvision==0.24.0 torchaudio==2.9.0; \
|
||||
torch=={torch_version} torchvision=={torchvision_version} torchaudio=={torchaudio_version}; \
|
||||
fi && \
|
||||
pip install --no-cache-dir --force-reinstall --no-deps \
|
||||
torch_npu==2.9.0.post2 && \
|
||||
torch_npu=={torch_npu_version} && \
|
||||
rm -rf /root/.cache/pip
|
||||
|
||||
# ---------- Remove CUDA-only dependencies pulled in by vllm (they cause missing libtorch_cuda.so errors on NPU) ----------
|
||||
@@ -83,7 +99,8 @@ ENV PYTHONPATH=/Megatron-LM:${PYTHONPATH}
|
||||
# install dependencies
|
||||
COPY requirements /var/modelscope
|
||||
|
||||
RUN pip uninstall ms-swift modelscope -y && pip install --no-cache-dir pip==23.* -U && \
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
pip uninstall ms-swift modelscope -y && pip install --no-cache-dir pip==23.* -U && \
|
||||
if [ "$INSTALL_MS_DEPS" = "True" ]; then \
|
||||
pip install --no-cache-dir omegaconf==2.0.6 && \
|
||||
pip install 'editdistance==0.8.1' && \
|
||||
@@ -109,9 +126,11 @@ fi
|
||||
ARG CUR_TIME={cur_time}
|
||||
RUN echo $CUR_TIME
|
||||
|
||||
RUN pip install --no-cache-dir --no-build-isolation OpenCC
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
pip install --no-cache-dir --no-build-isolation OpenCC
|
||||
|
||||
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
|
||||
pip install --no-cache-dir -U funasr scikit-learn && \
|
||||
pip install --no-cache-dir -U qwen_vl_utils qwen_omni_utils librosa 'timm>=0.9.0' transformers accelerate peft trl safetensors && \
|
||||
@@ -126,36 +145,20 @@ RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
pip install --no-cache-dir omegaconf==2.3.0 && \
|
||||
pip cache purge
|
||||
|
||||
# ---------- Reinstall triton-ascend for the selected CANN version ----------
|
||||
# ---------- Install training and evaluation dependencies ----------
|
||||
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
|
||||
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
|
||||
TORCH_DEVICE_BACKEND_AUTOLOAD=0 pip install --no-cache-dir "deepspeed<0.19" ray liger_kernel pre-commit -U && \
|
||||
pip cache purge
|
||||
|
||||
# ---------- Install triton-ascend ----------
|
||||
RUN set -eux; \
|
||||
export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}"; \
|
||||
pip uninstall -y triton || true; \
|
||||
pip uninstall -y triton-ascend || true; \
|
||||
case "${CANN_VERSION}" in \
|
||||
8.5.*) \
|
||||
pip install --no-cache-dir --force-reinstall triton-ascend==3.2.0; \
|
||||
;; \
|
||||
9.0.0) \
|
||||
PY_ABI="cp$(python -c 'import sys; print(f"{sys.version_info.major}{sys.version_info.minor}")')"; \
|
||||
case "${PY_ABI}" in \
|
||||
cp310|cp311|cp312|cp313) ;; \
|
||||
*) echo "Unsupported Python ABI for triton-ascend 3.2.1: ${PY_ABI}" >&2; exit 1 ;; \
|
||||
esac; \
|
||||
ARCH="$(uname -m)"; \
|
||||
case "${ARCH}" in \
|
||||
aarch64|x86_64) ;; \
|
||||
*) echo "Unsupported architecture for triton-ascend 3.2.1: ${ARCH}" >&2; exit 1 ;; \
|
||||
esac; \
|
||||
WHEEL_NAME="triton_ascend-3.2.1-${PY_ABI}-${PY_ABI}-manylinux_2_27_${ARCH}.manylinux_2_28_${ARCH}.whl"; \
|
||||
WHEEL_PATH="/tmp/${WHEEL_NAME}"; \
|
||||
curl -fL "https://gitcode.com/Ascend/triton-ascend/releases/download/v3.2.1/${WHEEL_NAME}" -o "${WHEEL_PATH}"; \
|
||||
pip install --no-cache-dir --force-reinstall "${WHEEL_PATH}"; \
|
||||
rm -f "${WHEEL_PATH}"; \
|
||||
;; \
|
||||
*) \
|
||||
echo "Unsupported CANN_VERSION for triton-ascend install: ${CANN_VERSION}" >&2; \
|
||||
exit 1; \
|
||||
;; \
|
||||
esac
|
||||
pip install --no-cache-dir --force-reinstall \
|
||||
triton-ascend=={triton_ascend_version} \
|
||||
--extra-index-url=https://triton-ascend.osinfra.cn/pypi/simple
|
||||
|
||||
RUN echo 'source /usr/local/Ascend/ascend-toolkit/set_env.sh' >> /root/.bashrc && \
|
||||
echo '[ -f /usr/local/Ascend/nnal/atb/set_env.sh ] && source /usr/local/Ascend/nnal/atb/set_env.sh' >> /root/.bashrc && \
|
||||
|
||||
@@ -3,6 +3,8 @@ FROM {base_image}
|
||||
ARG DEBIAN_FRONTEND=noninteractive
|
||||
ENV TZ=Asia/Shanghai
|
||||
ENV arch=x86_64
|
||||
# Build-time only (ARG does not persist into the image). Aliyun mirror can lag PyPI (~1h).
|
||||
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
|
||||
|
||||
COPY docker/scripts/modelscope_env_init.sh /usr/local/bin/ms_env_init.sh
|
||||
RUN apt-get update && \
|
||||
@@ -21,7 +23,9 @@ ARG IMAGE_TYPE={image_type}
|
||||
# install dependencies
|
||||
COPY requirements /var/modelscope
|
||||
|
||||
RUN pip uninstall ms-swift modelscope -y && pip --no-cache-dir install pip==23.* -U && \
|
||||
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
pip uninstall ms-swift modelscope -y && pip --no-cache-dir install pip==23.* -U && \
|
||||
if [ "$INSTALL_MS_DEPS" = "True" ]; then \
|
||||
pip --no-cache-dir install omegaconf==2.0.6 && \
|
||||
pip install 'editdistance==0.8.1' && \
|
||||
@@ -34,7 +38,7 @@ if [ "$INSTALL_MS_DEPS" = "True" ]; then \
|
||||
pip install --no-cache-dir 'scipy' && \
|
||||
pip install --no-cache-dir funtextprocessing typeguard==2.13.3 scikit-learn -f https://modelscope.oss-cn-beijing.aliyuncs.com/releases/repo.html && \
|
||||
pip install --no-cache-dir 'decord>=0.6.0' mpi4py paint_ldm ipykernel fasttext -f https://modelscope.oss-cn-beijing.aliyuncs.com/releases/repo.html && \
|
||||
pip install --no-cache-dir ipywidgets && \
|
||||
pip install --no-cache-dir ipywidgets jupyter_core nbconvert nbclient && \
|
||||
pip install --no-cache-dir 'blobfile>=1.0.5' && \
|
||||
pip uninstall MinDAEC -y && \
|
||||
pip install https://modelscope.oss-cn-beijing.aliyuncs.com/releases/dependencies/MinDAEC-0.0.2-py3-none-any.whl && \
|
||||
@@ -47,7 +51,9 @@ fi
|
||||
ARG CUR_TIME={cur_time}
|
||||
RUN echo $CUR_TIME
|
||||
|
||||
RUN bash /tmp/install.sh {version_args} && \
|
||||
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
bash /tmp/install.sh {version_args} && \
|
||||
pip install --no-cache-dir -U funasr scikit-learn && \
|
||||
pip install --no-cache-dir -U qwen_vl_utils qwen_omni_utils librosa timm transformers accelerate peft trl safetensors && \
|
||||
cd /tmp && GIT_LFS_SKIP_SMUDGE=1 git clone -b {swift_branch} --single-branch https://github.com/modelscope/ms-swift.git && \
|
||||
@@ -63,23 +69,17 @@ RUN bash /tmp/install.sh {version_args} && \
|
||||
pip install --no-cache-dir transformers diffusers 'timm>=0.9.0' && pip cache purge; \
|
||||
pip install --no-cache-dir omegaconf==2.3.0 && pip cache purge; \
|
||||
pip config set global.index-url https://mirrors.aliyun.com/pypi/simple && \
|
||||
pip config set global.extra-index-url https://pypi.org/simple && \
|
||||
pip config set install.trusted-host mirrors.aliyun.com && \
|
||||
cp /tmp/resources/ubuntu2204.aliyun /etc/apt/sources.list
|
||||
|
||||
|
||||
RUN if [ "$IMAGE_TYPE" = "gpu" ]; then \
|
||||
pip install --no-cache-dir math_verify "datasets<4.8.5" "gradio<5.33" "deepspeed<0.19" ray -U && \
|
||||
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
|
||||
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
|
||||
if [ "$IMAGE_TYPE" = "gpu" ]; then \
|
||||
pip install --no-cache-dir math_verify "gradio<5.33" "deepspeed<0.19" ray -U && \
|
||||
pip install --no-cache-dir mcore-bridge -i https://pypi.org/simple/ -U && \
|
||||
pip install --no-cache-dir pybind11 liger_kernel wandb swanlab nvitop pre-commit "transformers<5.15" "trl<1.0" "peft<0.21" huggingface-hub -U && \
|
||||
pip install git+https://github.com/NVIDIA/TransformerEngine.git@stable --no-build-isolation; \
|
||||
pip install git+https://github.com/deepseek-ai/DeepGEMM.git@v2.1.1.post3 --no-build-isolation; \
|
||||
pip install -U flash-linear-attention --no-build-isolation; \
|
||||
pip install -U git+https://github.com/Dao-AILab/causal-conv1d --no-build-isolation; \
|
||||
pip install git+https://github.com/Dao-AILab/fast-hadamard-transform --no-build-isolation; \
|
||||
pip install git+https://github.com/NVIDIA-NeMo/Emerging-Optimizers.git@v0.3.0; \
|
||||
mv /usr/local/lib/python3.12/site-packages/tilelang/lib/libcudart_stub.so /usr/local/lib/python3.12/site-packages/tilelang/lib/libcudart_stub.so.bak; \
|
||||
ln -s /usr/local/cuda-13.0/targets/x86_64-linux/lib/libcudart.so /usr/local/lib/python3.12/site-packages/tilelang/lib/libcudart_stub.so; \
|
||||
pip install --no-cache-dir liger_kernel wandb swanlab nvitop pre-commit "transformers" "trl<1.0" "peft<0.21" huggingface-hub -U && \
|
||||
pip install --no-cache-dir --no-build-isolation "transformer_engine[pytorch]==2.16.0"; \
|
||||
cd /tmp && GIT_LFS_SKIP_SMUDGE=1 git clone https://github.com/NVIDIA/apex && \
|
||||
cd apex && pip install -v --disable-pip-version-check --no-cache-dir --no-build-isolation --config-settings "--build-option=--cpp_ext" --config-settings "--build-option=--cuda_ext" ./ && \
|
||||
cd / && rm -fr /tmp/apex && pip cache purge; \
|
||||
|
||||
@@ -10,7 +10,8 @@ ms-swift Ascend images provide a ready-to-use ms-swift environment for Huawei As
|
||||
- Build template: `docker/Dockerfile.ascend`
|
||||
- Build entrypoint: `docker/build_image.py --image_type ascend`
|
||||
- Default base image: `quay.io/ascend/cann:8.5.1-a3-ubuntu22.04-py3.11`
|
||||
- Default output tag: `${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-<arch>`
|
||||
- Supported base OSes: Ubuntu and openEuler, selected from the CANN base-image tag
|
||||
- Default output tag: `${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-<arch>`
|
||||
- Ascend runtime environment is sourced from `/usr/local/Ascend/ascend-toolkit/set_env.sh`
|
||||
- If available, NNAL/ATB runtime is sourced from `/usr/local/Ascend/nnal/atb/set_env.sh`
|
||||
|
||||
@@ -22,45 +23,46 @@ The Ascend Dockerfile installs and configures:
|
||||
| --- | --- |
|
||||
| CANN | inherited from the selected `quay.io/ascend/cann` base image |
|
||||
| Python | inherited from the base image tag, for example `py3.11` |
|
||||
| PyTorch | `torch==2.9.0` |
|
||||
| torch-npu | `torch_npu==2.9.0.post2` |
|
||||
| torchvision / torchaudio | `torchvision==0.24.0`, `torchaudio==2.9.0` |
|
||||
| vLLM | source install from `vllm-project/vllm`, default branch `v0.18.0` |
|
||||
| vLLM Ascend | source install from `vllm-project/vllm-ascend`, default branch `v0.18.0` |
|
||||
| PyTorch | `torch==2.9.0` by default; configurable with `--torch_version` |
|
||||
| torch-npu | `torch_npu==2.9.0.post2` by default; configurable with `--torch_npu_version` |
|
||||
| torchvision / torchaudio | `torchvision==0.24.0`, `torchaudio==2.9.0` by default; pass both explicitly when overriding `--torch_version` |
|
||||
| vLLM | source install from `vllm-project/vllm`, default `0.18.0`; configurable with `--vllm_version` |
|
||||
| vLLM Ascend | source install from `vllm-project/vllm-ascend`, default `0.18.0`; configurable with `--vllm_ascend_version` |
|
||||
| Megatron-LM | source checkout, default branch `v0.15.3` |
|
||||
| MindSpeed | source checkout, default branch `core_r0.15.3` |
|
||||
| mcore-bridge | source checkout from `modelscope/mcore-bridge` |
|
||||
| mcore-bridge | latest release from PyPI |
|
||||
| ms-swift | source checkout from `modelscope/ms-swift`, default branch `main` |
|
||||
| ModelScope | source checkout from `modelscope/modelscope`, default branch `master` |
|
||||
| triton-ascend | `3.2.0` for CANN `8.5.*`; local wheel install of `3.2.1` for CANN `9.0.0` |
|
||||
| triton-ascend | CANN `8.5.*` defaults to `3.2.0`; CANN `9.0.*` defaults to `3.2.1`; configurable with `--triton_ascend_version` and installed from the Triton Ascend PyPI index |
|
||||
|
||||
## Supported Tag Format
|
||||
|
||||
Images built by `docker/build_image.py --image_type ascend` use this tag format:
|
||||
|
||||
```text
|
||||
${DOCKER_REGISTRY}:<swift-branch>-<atlas-hardware>-<python-tag>-<cann-version-tag>-<os-tag>-<arch>
|
||||
${DOCKER_REGISTRY}:<swift-branch>-<cann-version-tag>-torch_npu<torch-npu-version>-<atlas-hardware>-<os-tag>-<python-tag>-<arch>
|
||||
```
|
||||
|
||||
| Field | Example | Description |
|
||||
| --- | --- | --- |
|
||||
| `swift-branch` | `main` | ms-swift branch used during image build |
|
||||
| `atlas-hardware` | `A2`, `A3`, `300I`, `A5` | Derived from `--soc_version` |
|
||||
| `python-tag` | `py311` | Derived from `--python_version` |
|
||||
| `cann-version-tag` | `CANN8.5.1`, `CANN9.0.0` | Parsed from the CANN base image tag |
|
||||
| `os-tag` | `ubuntu22.04` | Parsed from the CANN base image tag |
|
||||
| `arch` | `arm`, `x86` | Derived from host architecture or `--arch` |
|
||||
| `cann-version-tag` | `cann8.5.1`, `cann9.0.0` | Parsed from the CANN base image tag |
|
||||
| `torch-npu-version` | `2.9.0.post2` | From `--torch_npu_version`; defaults to `2.9.0.post2` |
|
||||
| `atlas-hardware` | `a2`, `a3`, `300i`, `a5` | Derived from `--soc_version` |
|
||||
| `os-tag` | `ubuntu22.04`, `openeuler24.03` | Parsed from the CANN base-image tag; prevents tags for different OSes from colliding |
|
||||
| `python-tag` | `py3.11` | Parsed from the CANN base image tag |
|
||||
| `arch` | `aarch64`, `x86_64` | Derived from host architecture or `--arch` |
|
||||
|
||||
Default example on an ARM64 host:
|
||||
|
||||
```text
|
||||
${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-arm
|
||||
${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-aarch64
|
||||
```
|
||||
|
||||
A2 / CANN 9.0.0 example:
|
||||
|
||||
```text
|
||||
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm
|
||||
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64
|
||||
```
|
||||
|
||||
## Build Locally
|
||||
@@ -85,6 +87,39 @@ python docker/build_image.py \
|
||||
--soc_version ascend910b1
|
||||
```
|
||||
|
||||
Build an openEuler image. The system-dependency layer automatically uses `yum`; Ubuntu images continue to use `apt-get`.
|
||||
|
||||
```bash
|
||||
python docker/build_image.py \
|
||||
--image_type ascend \
|
||||
--base_image quay.io/ascend/cann:8.5.1-a3-openeuler24.03-py3.11 \
|
||||
--soc_version ascend910_9391
|
||||
```
|
||||
|
||||
Override the PyTorch stack. `--torch_version` must match the base version of
|
||||
`--torch_npu_version`; when overriding PyTorch, pass its matching torchvision
|
||||
and torchaudio versions explicitly.
|
||||
|
||||
```bash
|
||||
python docker/build_image.py \
|
||||
--image_type ascend \
|
||||
--torch_version 2.9.0 \
|
||||
--torch_npu_version 2.9.0.post2 \
|
||||
--torchvision_version 0.24.0 \
|
||||
--torchaudio_version 2.9.0
|
||||
```
|
||||
|
||||
Override the vLLM stack or triton-ascend. The vLLM version arguments select
|
||||
the matching Git tag, for example `0.18.0` selects `v0.18.0`.
|
||||
|
||||
```bash
|
||||
python docker/build_image.py \
|
||||
--image_type ascend \
|
||||
--vllm_version 0.18.0 \
|
||||
--vllm_ascend_version 0.18.0 \
|
||||
--triton_ascend_version 3.2.1
|
||||
```
|
||||
|
||||
Override Megatron or MindSpeed source branches when needed:
|
||||
|
||||
```bash
|
||||
@@ -94,11 +129,11 @@ python docker/build_image.py \
|
||||
--mindspeed_branch core_r0.15.3
|
||||
```
|
||||
|
||||
For slow networks, Linux hosts can use Docker host networking after the root `Dockerfile` is generated:
|
||||
To run the rendered Dockerfile manually, use:
|
||||
|
||||
```bash
|
||||
docker build --network host \
|
||||
-t ${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
|
||||
docker build \
|
||||
-t ${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
|
||||
-f Dockerfile .
|
||||
```
|
||||
|
||||
@@ -119,7 +154,7 @@ docker run --rm -it \
|
||||
-v /usr/local/Ascend/driver/version.info:/usr/local/Ascend/driver/version.info \
|
||||
-v /etc/ascend_install.info:/etc/ascend_install.info \
|
||||
-v /mnt/workspace:/mnt/workspace \
|
||||
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
|
||||
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
|
||||
bash
|
||||
```
|
||||
|
||||
@@ -147,7 +182,8 @@ pip show ms-swift modelscope torch-npu triton-ascend
|
||||
## Notes
|
||||
|
||||
- CANN, firmware, and driver versions must be compatible with each other.
|
||||
- CANN `8.5.*` and CANN `9.0.0` use different `triton-ascend` install paths in this Dockerfile.
|
||||
- Ubuntu base images install system dependencies through `apt-get`; openEuler base images install the corresponding RPM packages through `yum`.
|
||||
- `triton-ascend` is installed from `https://triton-ascend.osinfra.cn/pypi/simple`; select a version compatible with the chosen CANN, Python, and architecture.
|
||||
- The image is intended for Ascend NPU ms-swift workflows. CUDA-only packages pulled in by dependencies are removed when they conflict with NPU runtime libraries.
|
||||
- Use a fixed image tag for production jobs instead of relying on a moving branch name.
|
||||
|
||||
|
||||
@@ -10,7 +10,8 @@ ms-swift Ascend 镜像面向华为昇腾 Atlas NPU,提供可直接使用的 ms
|
||||
- 构建模板:`docker/Dockerfile.ascend`
|
||||
- 构建入口:`docker/build_image.py --image_type ascend`
|
||||
- 默认基础镜像:`quay.io/ascend/cann:8.5.1-a3-ubuntu22.04-py3.11`
|
||||
- 默认输出 tag:`${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-<arch>`
|
||||
- 支持的基础 OS:Ubuntu 和 openEuler,由 CANN 基础镜像 tag 选择
|
||||
- 默认输出 tag:`${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-<arch>`
|
||||
- Ascend runtime 环境来自 `/usr/local/Ascend/ascend-toolkit/set_env.sh`
|
||||
- 如果镜像内存在 NNAL/ATB,则会加载 `/usr/local/Ascend/nnal/atb/set_env.sh`
|
||||
|
||||
@@ -22,45 +23,46 @@ Ascend Dockerfile 会安装和配置:
|
||||
| --- | --- |
|
||||
| CANN | 继承自选定的 `quay.io/ascend/cann` 基础镜像 |
|
||||
| Python | 继承自基础镜像 tag,例如 `py3.11` |
|
||||
| PyTorch | `torch==2.9.0` |
|
||||
| torch-npu | `torch_npu==2.9.0.post2` |
|
||||
| torchvision / torchaudio | `torchvision==0.24.0`,`torchaudio==2.9.0` |
|
||||
| vLLM | 从 `vllm-project/vllm` 源码安装,默认分支 `v0.18.0` |
|
||||
| vLLM Ascend | 从 `vllm-project/vllm-ascend` 源码安装,默认分支 `v0.18.0` |
|
||||
| PyTorch | 默认 `torch==2.9.0`;可通过 `--torch_version` 配置 |
|
||||
| torch-npu | 默认 `torch_npu==2.9.0.post2`;可通过 `--torch_npu_version` 配置 |
|
||||
| torchvision / torchaudio | 默认 `torchvision==0.24.0`、`torchaudio==2.9.0`;覆盖 `--torch_version` 时必须同时显式传入两者 |
|
||||
| vLLM | 从 `vllm-project/vllm` 源码安装,默认 `0.18.0`;可通过 `--vllm_version` 配置 |
|
||||
| vLLM Ascend | 从 `vllm-project/vllm-ascend` 源码安装,默认 `0.18.0`;可通过 `--vllm_ascend_version` 配置 |
|
||||
| Megatron-LM | 源码 checkout,默认分支 `v0.15.3` |
|
||||
| MindSpeed | 源码 checkout,默认分支 `core_r0.15.3` |
|
||||
| mcore-bridge | 来自 `modelscope/mcore-bridge` 的源码 checkout |
|
||||
| mcore-bridge | PyPI 上的最新发布版 |
|
||||
| ms-swift | 来自 `modelscope/ms-swift` 的源码 checkout,默认分支 `main` |
|
||||
| ModelScope | 来自 `modelscope/modelscope` 的源码 checkout,默认分支 `master` |
|
||||
| triton-ascend | CANN `8.5.*` 安装 `3.2.0`;CANN `9.0.0` 下载并本地安装 `3.2.1` wheel |
|
||||
| triton-ascend | CANN `8.5.*` 默认 `3.2.0`;CANN `9.0.*` 默认 `3.2.1`;可通过 `--triton_ascend_version` 配置,并从 Triton Ascend PyPI 源安装 |
|
||||
|
||||
## 支持的 Tag 格式
|
||||
|
||||
通过 `docker/build_image.py --image_type ascend` 构建的镜像使用以下 tag 格式:
|
||||
|
||||
```text
|
||||
${DOCKER_REGISTRY}:<swift-branch>-<atlas-hardware>-<python-tag>-<cann-version-tag>-<os-tag>-<arch>
|
||||
${DOCKER_REGISTRY}:<swift-branch>-<cann-version-tag>-torch_npu<torch-npu-version>-<atlas-hardware>-<os-tag>-<python-tag>-<arch>
|
||||
```
|
||||
|
||||
| 字段 | 示例 | 说明 |
|
||||
| --- | --- | --- |
|
||||
| `swift-branch` | `main` | 构建镜像时使用的 ms-swift 分支 |
|
||||
| `atlas-hardware` | `A2`、`A3`、`300I`、`A5` | 从 `--soc_version` 推导 |
|
||||
| `python-tag` | `py311` | 从 `--python_version` 推导 |
|
||||
| `cann-version-tag` | `CANN8.5.1`、`CANN9.0.0` | 从 CANN 基础镜像 tag 解析 |
|
||||
| `os-tag` | `ubuntu22.04` | 从 CANN 基础镜像 tag 解析 |
|
||||
| `arch` | `arm`、`x86` | 从宿主机架构或 `--arch` 推导 |
|
||||
| `cann-version-tag` | `cann8.5.1`、`cann9.0.0` | 从 CANN 基础镜像 tag 解析 |
|
||||
| `torch-npu-version` | `2.9.0.post2` | 来自 `--torch_npu_version`,默认 `2.9.0.post2` |
|
||||
| `atlas-hardware` | `a2`、`a3`、`300i`、`a5` | 从 `--soc_version` 推导 |
|
||||
| `os-tag` | `ubuntu22.04`、`openeuler24.03` | 从 CANN 基础镜像 tag 解析;避免不同 OS 的镜像 tag 冲突 |
|
||||
| `python-tag` | `py3.11` | 从 CANN 基础镜像 tag 解析 |
|
||||
| `arch` | `aarch64`、`x86_64` | 从宿主机架构或 `--arch` 推导 |
|
||||
|
||||
ARM64 宿主机上的默认示例:
|
||||
|
||||
```text
|
||||
${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-arm
|
||||
${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-aarch64
|
||||
```
|
||||
|
||||
A2 / CANN 9.0.0 示例:
|
||||
|
||||
```text
|
||||
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm
|
||||
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64
|
||||
```
|
||||
|
||||
## 本地构建
|
||||
@@ -85,6 +87,36 @@ python docker/build_image.py \
|
||||
--soc_version ascend910b1
|
||||
```
|
||||
|
||||
构建 openEuler 镜像。系统依赖层会自动使用 `yum`,Ubuntu 镜像继续使用 `apt-get`:
|
||||
|
||||
```bash
|
||||
python docker/build_image.py \
|
||||
--image_type ascend \
|
||||
--base_image quay.io/ascend/cann:8.5.1-a3-openeuler24.03-py3.11 \
|
||||
--soc_version ascend910_9391
|
||||
```
|
||||
|
||||
覆盖 PyTorch 版本组。`--torch_version` 必须与 `--torch_npu_version` 的基础版本一致;覆盖 PyTorch 时,必须显式传入匹配的 torchvision 和 torchaudio 版本:
|
||||
|
||||
```bash
|
||||
python docker/build_image.py \
|
||||
--image_type ascend \
|
||||
--torch_version 2.9.0 \
|
||||
--torch_npu_version 2.9.0.post2 \
|
||||
--torchvision_version 0.24.0 \
|
||||
--torchaudio_version 2.9.0
|
||||
```
|
||||
|
||||
覆盖 vLLM 版本组或 triton-ascend。vLLM 版本参数会选择对应 Git tag,例如 `0.18.0` 会选择 `v0.18.0`。
|
||||
|
||||
```bash
|
||||
python docker/build_image.py \
|
||||
--image_type ascend \
|
||||
--vllm_version 0.18.0 \
|
||||
--vllm_ascend_version 0.18.0 \
|
||||
--triton_ascend_version 3.2.1
|
||||
```
|
||||
|
||||
需要时可以覆盖 Megatron 或 MindSpeed 源码分支:
|
||||
|
||||
```bash
|
||||
@@ -94,11 +126,11 @@ python docker/build_image.py \
|
||||
--mindspeed_branch core_r0.15.3
|
||||
```
|
||||
|
||||
如果构建时网络较慢,Linux 宿主机可以在根目录 `Dockerfile` 生成后使用 host network 构建:
|
||||
如需手工构建生成后的根目录 `Dockerfile`,可使用:
|
||||
|
||||
```bash
|
||||
docker build --network host \
|
||||
-t ${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
|
||||
docker build \
|
||||
-t ${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
|
||||
-f Dockerfile .
|
||||
```
|
||||
|
||||
@@ -119,7 +151,7 @@ docker run --rm -it \
|
||||
-v /usr/local/Ascend/driver/version.info:/usr/local/Ascend/driver/version.info \
|
||||
-v /etc/ascend_install.info:/etc/ascend_install.info \
|
||||
-v /mnt/workspace:/mnt/workspace \
|
||||
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
|
||||
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
|
||||
bash
|
||||
```
|
||||
|
||||
@@ -147,7 +179,8 @@ pip show ms-swift modelscope torch-npu triton-ascend
|
||||
## 注意事项
|
||||
|
||||
- CANN、firmware 和 driver 版本必须互相兼容。
|
||||
- 这个 Dockerfile 对 CANN `8.5.*` 和 CANN `9.0.0` 使用不同的 `triton-ascend` 安装路径。
|
||||
- Ubuntu 基础镜像通过 `apt-get` 安装系统依赖;openEuler 基础镜像通过 `yum` 安装对应 RPM 包。
|
||||
- `triton-ascend` 从 `https://triton-ascend.osinfra.cn/pypi/simple` 安装;请选择与 CANN、Python 和架构兼容的版本。
|
||||
- 该镜像面向 Ascend NPU 上的 ms-swift 工作流。依赖安装过程中引入且与 NPU runtime 冲突的 CUDA-only 包会被移除。
|
||||
- 生产任务建议使用固定镜像 tag,不要依赖浮动分支名。
|
||||
|
||||
|
||||
@@ -3,14 +3,27 @@ import os
|
||||
import platform
|
||||
import re
|
||||
import subprocess
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from copy import copy
|
||||
from datetime import datetime
|
||||
from typing import Any
|
||||
from typing import Any, List, Optional
|
||||
|
||||
import json
|
||||
|
||||
docker_registry = os.environ['DOCKER_REGISTRY']
|
||||
assert docker_registry, 'You must pass a valid DOCKER_REGISTRY'
|
||||
timestamp = datetime.now()
|
||||
formatted_time = timestamp.strftime('%Y%m%d%H%M%S')
|
||||
VLLM_ROCM_REPO = 'vllm/vllm-openai-rocm'
|
||||
_FLOATING_ROCM_TAGS = frozenset({
|
||||
'latest',
|
||||
'latest-base',
|
||||
'nightly',
|
||||
'base-nightly',
|
||||
})
|
||||
_VERSION_TAG_PATTERN = re.compile(r'^v\d+(?:\.\d+)*$')
|
||||
_NIGHTLY_HASH_PATTERN = re.compile(r'^(?:base-)?nightly-[0-9a-f]{7,40}$')
|
||||
|
||||
|
||||
class Builder:
|
||||
@@ -359,7 +372,8 @@ class StableGPUImageBuilder(Builder):
|
||||
extra_content = extra_content.replace('{python_version}',
|
||||
self.args.python_version)
|
||||
extra_content += """
|
||||
RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
RUN export PIP_EXTRA_INDEX_URL=https://pypi.org/simple && \
|
||||
pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
"""
|
||||
version_args = (
|
||||
f'{self.args.torch_version} {self.args.torchvision_version} {self.args.torchaudio_version} '
|
||||
@@ -434,7 +448,8 @@ class LatestGPUImageBuilder(StableGPUImageBuilder):
|
||||
extra_content = extra_content.replace('{python_version}',
|
||||
self.args.python_version)
|
||||
extra_content += """
|
||||
RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
RUN export PIP_EXTRA_INDEX_URL=https://pypi.org/simple && \
|
||||
pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
"""
|
||||
version_args = (
|
||||
f'{self.args.torch_version} {self.args.torchvision_version} {self.args.torchaudio_version} '
|
||||
@@ -481,22 +496,399 @@ RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
return self.run_cmd('docker', 'push', image_tag2)
|
||||
|
||||
|
||||
class AmdImageBuilder(Builder):
|
||||
"""Build ModelScope image on top of vllm/vllm-openai-rocm."""
|
||||
|
||||
@staticmethod
|
||||
def _is_specific_release_tag(tag: str) -> bool:
|
||||
tag = tag.strip()
|
||||
if not tag or tag.lower() in _FLOATING_ROCM_TAGS:
|
||||
return False
|
||||
if tag.endswith('-base'):
|
||||
return False
|
||||
if _NIGHTLY_HASH_PATTERN.fullmatch(tag):
|
||||
return False
|
||||
return bool(_VERSION_TAG_PATTERN.fullmatch(tag))
|
||||
|
||||
@staticmethod
|
||||
def _image_digest(tag_info: dict) -> Optional[str]:
|
||||
digest = tag_info.get('digest')
|
||||
if digest:
|
||||
return digest
|
||||
for image in tag_info.get('images') or []:
|
||||
digest = image.get('digest')
|
||||
if digest:
|
||||
return digest
|
||||
return None
|
||||
|
||||
@classmethod
|
||||
def _fetch_rocm_tags(cls, page_size: int = 100) -> List[dict]:
|
||||
tags: List[dict] = []
|
||||
url = (f'https://hub.docker.com/v2/repositories/{VLLM_ROCM_REPO}/tags'
|
||||
f'?page_size={page_size}&ordering=-last_updated')
|
||||
while url:
|
||||
req = urllib.request.Request(
|
||||
url, headers={'User-Agent': 'modelscope-docker-builder'})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=60) as resp:
|
||||
payload = json.load(resp)
|
||||
except (urllib.error.URLError, json.JSONDecodeError) as exc:
|
||||
raise RuntimeError(
|
||||
f'Failed to query Docker Hub tags for {VLLM_ROCM_REPO}: '
|
||||
f'{exc}') from exc
|
||||
tags.extend(payload.get('results') or [])
|
||||
url = payload.get('next')
|
||||
# Only scan the first few pages; release tags are near the top.
|
||||
if len(tags) >= 300:
|
||||
break
|
||||
if not tags:
|
||||
raise RuntimeError(
|
||||
f'No tags returned from Docker Hub for {VLLM_ROCM_REPO}')
|
||||
return tags
|
||||
|
||||
@classmethod
|
||||
def resolve_latest_rocm_tag(cls) -> str:
|
||||
"""Resolve the newest concrete release tag for vllm-openai-rocm.
|
||||
|
||||
Preference order:
|
||||
1. Semver tag (vX.Y.Z) that shares digest with floating ``latest``
|
||||
2. Newest semver tag by Docker Hub ``last_updated``
|
||||
"""
|
||||
tags = cls._fetch_rocm_tags()
|
||||
by_name = {item['name']: item for item in tags if item.get('name')}
|
||||
release_tags = [
|
||||
item for item in tags
|
||||
if cls._is_specific_release_tag(item.get('name', ''))
|
||||
]
|
||||
latest_info = by_name.get('latest')
|
||||
latest_digest = cls._image_digest(latest_info) if latest_info else None
|
||||
if latest_digest:
|
||||
matched = [
|
||||
item for item in release_tags
|
||||
if cls._image_digest(item) == latest_digest
|
||||
]
|
||||
if matched:
|
||||
# Prefer the first match in last_updated order from API.
|
||||
chosen = matched[0]['name']
|
||||
print(
|
||||
f'Resolved {VLLM_ROCM_REPO} latest digest to release tag: '
|
||||
f'{chosen}')
|
||||
return chosen
|
||||
|
||||
if not release_tags:
|
||||
raise RuntimeError(
|
||||
f'No concrete release tags found for {VLLM_ROCM_REPO}')
|
||||
chosen = release_tags[0]['name']
|
||||
print(f'Resolved newest {VLLM_ROCM_REPO} release tag: {chosen}')
|
||||
return chosen
|
||||
|
||||
def init_args(self, args: Any) -> Any:
|
||||
# Auto-discover from Docker Hub unless an explicit override is given.
|
||||
override = getattr(args, 'base_image_tag', None)
|
||||
if override and str(override).strip() and str(
|
||||
override).strip().lower() not in {'auto', 'latest'}:
|
||||
args.base_image_tag = str(override).strip()
|
||||
if not self._is_specific_release_tag(args.base_image_tag):
|
||||
raise ValueError(
|
||||
'base_image_tag override must be a concrete release tag '
|
||||
f'(e.g. v0.25.1), got: {args.base_image_tag}')
|
||||
print(f'Using override AMD ROCm base image tag: '
|
||||
f'{args.base_image_tag}')
|
||||
else:
|
||||
args.base_image_tag = self.resolve_latest_rocm_tag()
|
||||
if not args.base_image:
|
||||
args.base_image = f'{VLLM_ROCM_REPO}:{args.base_image_tag}'
|
||||
if not args.cuda_version:
|
||||
args.cuda_version = '0.0.0'
|
||||
return args
|
||||
|
||||
@staticmethod
|
||||
def _sanitize_tag(tag: str) -> str:
|
||||
return re.sub(r'[^A-Za-z0-9._-]+', '-', tag)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_version(version: str) -> str:
|
||||
version = version.strip().lstrip('vV')
|
||||
version = version.split('+')[0].split(' ')[0]
|
||||
return re.sub(r'[^0-9A-Za-z._-]+', '', version)
|
||||
|
||||
@staticmethod
|
||||
def _python_tag_from_version(version: str) -> str:
|
||||
parts = version.strip().split('.')
|
||||
if len(parts) >= 2 and parts[0].isdigit() and parts[1].isdigit():
|
||||
return f'py{parts[0]}{parts[1]}'
|
||||
return f'py{re.sub(r"[^0-9]", "", version)}'
|
||||
|
||||
@classmethod
|
||||
def _run_capture(cls, *cmd: str) -> subprocess.CompletedProcess:
|
||||
return subprocess.run(
|
||||
list(cmd), capture_output=True, text=True, check=False)
|
||||
|
||||
@classmethod
|
||||
def _probe_via_entrypoint(cls, base_image: str) -> dict:
|
||||
"""Read versions with docker run --entrypoint (no GPU required)."""
|
||||
# Keep this script compact: it runs inside the base image via python -c.
|
||||
script = (
|
||||
'import json,os,pathlib,subprocess,sys\n'
|
||||
'info={"python":"%d.%d.%d"%sys.version_info[:3]}\n'
|
||||
'try:\n'
|
||||
' import torch\n'
|
||||
' info["torch"]=torch.__version__\n'
|
||||
' hip=getattr(torch.version,"hip",None)\n'
|
||||
' if hip: info["torch_hip"]=hip\n'
|
||||
'except Exception as e:\n'
|
||||
' info["torch_error"]=str(e)\n'
|
||||
'for p in ("/opt/rocm/.info/version","/opt/rocm/.info/version-dev"):\n'
|
||||
' f=pathlib.Path(p)\n'
|
||||
' if f.is_file():\n'
|
||||
' info["rocm_file"]=f.read_text().strip().splitlines()[0]\n'
|
||||
' break\n'
|
||||
'for k in ("ROCM_VERSION","HIP_VERSION","TORCH_VERSION"):\n'
|
||||
' if os.environ.get(k): info[k.lower()]=os.environ[k]\n'
|
||||
'def _dpkg_ver(*names):\n'
|
||||
' for n in names:\n'
|
||||
' try:\n'
|
||||
' r=subprocess.run(["dpkg-query","-W","-f=${Version}",n],'
|
||||
'capture_output=True,text=True)\n'
|
||||
' if r.returncode==0 and r.stdout.strip():\n'
|
||||
' return r.stdout.strip()\n'
|
||||
' except Exception:\n'
|
||||
' pass\n'
|
||||
' return None\n'
|
||||
'def _dpkg_scan(prefixes):\n'
|
||||
' try:\n'
|
||||
' r=subprocess.run(["dpkg-query","-W","-f=${Package}\\t${Version}\\n"],'
|
||||
'capture_output=True,text=True)\n'
|
||||
' except Exception:\n'
|
||||
' return {}\n'
|
||||
' found={}\n'
|
||||
' for line in (r.stdout or "").splitlines():\n'
|
||||
' if "\\t" not in line: continue\n'
|
||||
' pkg,ver=line.split("\\t",1)\n'
|
||||
' for pref in prefixes:\n'
|
||||
' if pkg==pref or pkg.startswith(pref+"-"):\n'
|
||||
' found.setdefault(pref,ver)\n'
|
||||
' return found\n'
|
||||
'pkgs=_dpkg_scan(("rccl","miopen"))\n'
|
||||
'info["system.library.rccl"]=_dpkg_ver("rccl") or pkgs.get("rccl")\n'
|
||||
'info["system.library.miopen"]=('
|
||||
'_dpkg_ver("miopen-hip","miopen") or pkgs.get("miopen"))\n'
|
||||
'print(json.dumps(info))\n')
|
||||
for py in ('python3', 'python'):
|
||||
result = cls._run_capture('docker', 'run', '--rm', '--network',
|
||||
'none', '--entrypoint', py, base_image,
|
||||
'-c', script)
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
try:
|
||||
return json.loads(result.stdout.strip().splitlines()[-1])
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return {}
|
||||
|
||||
@classmethod
|
||||
def _probe_via_history(cls, base_image: str) -> dict:
|
||||
"""Parse build ARGs from docker history (no container start)."""
|
||||
result = cls._run_capture('docker', 'history', '--no-trunc',
|
||||
'--format', '{{.CreatedBy}}', base_image)
|
||||
if result.returncode != 0:
|
||||
return {}
|
||||
text = result.stdout
|
||||
info = {}
|
||||
for key, pattern in (
|
||||
('rocm', r'ROCM_VERSION=([0-9]+(?:\.[0-9]+)*)'),
|
||||
('python', r'PYTHON_VERSION=([0-9]+(?:\.[0-9]+)*)'),
|
||||
('ubuntu',
|
||||
r'org\.opencontainers\.image\.version=([0-9]+(?:\.[0-9]+)*)'),
|
||||
):
|
||||
matches = re.findall(pattern, text)
|
||||
if matches:
|
||||
# docker history lists newest layers first.
|
||||
info[key] = matches[0]
|
||||
return info
|
||||
|
||||
@classmethod
|
||||
def _probe_via_create_cp(cls, base_image: str) -> dict:
|
||||
"""Copy version files out of a created (not started) container."""
|
||||
import tempfile
|
||||
create = cls._run_capture('docker', 'create', base_image)
|
||||
if create.returncode != 0:
|
||||
return {}
|
||||
cid = create.stdout.strip()
|
||||
info = {}
|
||||
try:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
dest = os.path.join(tmp, 'version')
|
||||
for src in ('/opt/rocm/.info/version',
|
||||
'/opt/rocm/.info/version-dev'):
|
||||
result = cls._run_capture('docker', 'cp', f'{cid}:{src}',
|
||||
dest)
|
||||
if result.returncode == 0 and os.path.isfile(dest):
|
||||
with open(dest, 'r', encoding='utf-8') as f:
|
||||
line = f.read().strip().splitlines()
|
||||
if line:
|
||||
info['rocm_file'] = line[0].strip()
|
||||
break
|
||||
finally:
|
||||
cls._run_capture('docker', 'rm', '-f', cid)
|
||||
return info
|
||||
|
||||
@classmethod
|
||||
def probe_base_image_versions(cls, base_image: str) -> dict:
|
||||
"""Discover rocm/python/torch without needing AMD GPU.
|
||||
|
||||
Methods (in order):
|
||||
1. docker run --entrypoint python -c ... (CPU-only, no --device)
|
||||
2. docker history --no-trunc parse ROCM_VERSION/PYTHON_VERSION
|
||||
3. docker create + docker cp /opt/rocm/.info/version
|
||||
"""
|
||||
probed = {}
|
||||
entry = cls._probe_via_entrypoint(base_image)
|
||||
history = cls._probe_via_history(base_image)
|
||||
copied = cls._probe_via_create_cp(base_image)
|
||||
probed.update(history)
|
||||
probed.update(copied)
|
||||
probed.update(entry)
|
||||
|
||||
rocm = (
|
||||
probed.get('rocm_file') or probed.get('rocm_version')
|
||||
or probed.get('rocm') or probed.get('torch_hip')
|
||||
or probed.get('hip_version'))
|
||||
python_ver = probed.get('python')
|
||||
torch_ver = probed.get('torch') or probed.get('torch_version')
|
||||
ubuntu_ver = probed.get('ubuntu')
|
||||
# Keep dpkg package versions as-is (may contain '~', e.g. 2.27.7.70201-81~22.04).
|
||||
rccl_ver = probed.get('system.library.rccl')
|
||||
miopen_ver = probed.get('system.library.miopen')
|
||||
|
||||
versions = {
|
||||
'rocm': cls._normalize_version(rocm) if rocm else None,
|
||||
'python':
|
||||
cls._normalize_version(python_ver) if python_ver else None,
|
||||
'torch': cls._normalize_version(torch_ver) if torch_ver else None,
|
||||
'ubuntu':
|
||||
cls._normalize_version(ubuntu_ver) if ubuntu_ver else None,
|
||||
'system.library.rccl': rccl_ver or None,
|
||||
'system.library.miopen': miopen_ver or None,
|
||||
}
|
||||
print('Probed AMD base image versions:')
|
||||
for key, value in versions.items():
|
||||
print(f' {key}: {value or "unknown"}')
|
||||
return versions
|
||||
|
||||
def generate_dockerfile(self) -> str:
|
||||
with open('docker/Dockerfile.amd', 'r') as f:
|
||||
content = f.read()
|
||||
content = content.replace('{base_image}', self.args.base_image)
|
||||
content = content.replace('{base_image_tag}', self.args.base_image_tag)
|
||||
content = content.replace('{modelscope_branch}',
|
||||
self.args.modelscope_branch)
|
||||
content = content.replace('{cur_time}', formatted_time)
|
||||
return content
|
||||
|
||||
def image(self) -> str:
|
||||
ubuntu = getattr(self.args, 'amd_ubuntu_version',
|
||||
None) or self.args.ubuntu_version
|
||||
rocm = getattr(self.args, 'amd_rocm_version', None)
|
||||
py_tag = getattr(self.args, 'amd_python_tag', None) or getattr(
|
||||
self.args, 'python_tag', None)
|
||||
torch = getattr(self.args, 'amd_torch_version', None)
|
||||
if not (rocm and py_tag and torch):
|
||||
raise RuntimeError(
|
||||
'AMD image tag requires probed rocm/python/torch versions. '
|
||||
f'Got rocm={rocm}, python={py_tag}, torch={torch}')
|
||||
return (f'{docker_registry}:ubuntu{ubuntu}-rocm{rocm}-{py_tag}-'
|
||||
f'torch{torch}-{self.args.modelscope_version}-test')
|
||||
|
||||
def _log_base_image_info(self) -> int:
|
||||
base_image = self.args.base_image
|
||||
print('=' * 60)
|
||||
print(f'AMD ROCm base image: {base_image}')
|
||||
print(f'AMD ROCm base image tag: {self.args.base_image_tag}')
|
||||
print('=' * 60)
|
||||
ret = self.run_cmd('docker', 'pull', base_image)
|
||||
if ret != 0:
|
||||
return ret
|
||||
result = self._run_capture(
|
||||
'docker', 'image', 'inspect', base_image,
|
||||
'--format={{.Id}} {{if index .RepoDigests 0}}'
|
||||
'{{index .RepoDigests 0}}{{else}}local-only{{end}}')
|
||||
if result.returncode == 0:
|
||||
print(f'AMD base image resolved: {result.stdout.strip()}')
|
||||
else:
|
||||
print(f'AMD base image inspect warning: {result.stderr.strip()}')
|
||||
|
||||
versions = self.probe_base_image_versions(base_image)
|
||||
if not versions.get('rocm') or not versions.get(
|
||||
'python') or not versions.get('torch'):
|
||||
print('ERROR: failed to probe rocm/python/torch from base image')
|
||||
return 1
|
||||
self.args.amd_rocm_version = versions['rocm']
|
||||
self.args.amd_torch_version = versions['torch']
|
||||
self.args.amd_python_tag = self._python_tag_from_version(
|
||||
versions['python'])
|
||||
if versions.get('ubuntu'):
|
||||
self.args.amd_ubuntu_version = versions['ubuntu']
|
||||
else:
|
||||
self.args.amd_ubuntu_version = self.args.ubuntu_version
|
||||
print(f'AMD output image tag will be: {self.image()}')
|
||||
print('=' * 60)
|
||||
return 0
|
||||
|
||||
def build(self) -> int:
|
||||
ret = self._log_base_image_info()
|
||||
if ret != 0:
|
||||
return ret
|
||||
return self.run_cmd('docker', 'build', '-t', self.image(), '-f',
|
||||
'Dockerfile', '.')
|
||||
|
||||
def push(self):
|
||||
image_name = self.image()
|
||||
ret = self.run_cmd('docker', 'push', image_name)
|
||||
if ret != 0:
|
||||
return ret
|
||||
ubuntu = self.args.amd_ubuntu_version
|
||||
rocm = self.args.amd_rocm_version
|
||||
py_tag = self.args.amd_python_tag
|
||||
torch = self.args.amd_torch_version
|
||||
image_tag2 = (f'{docker_registry}:ubuntu{ubuntu}-rocm{rocm}-{py_tag}-'
|
||||
f'torch{torch}-{self.args.modelscope_version}-'
|
||||
f'{formatted_time}-test')
|
||||
ret = self.run_cmd('docker', 'tag', image_name, image_tag2)
|
||||
if ret != 0:
|
||||
return ret
|
||||
print(f'AMD image timestamp tag: {image_tag2}')
|
||||
return self.run_cmd('docker', 'push', image_tag2)
|
||||
|
||||
|
||||
class AscendImageBuilder(StableGPUImageBuilder):
|
||||
|
||||
_DEFAULT_TORCH_VERSION = '2.9.0'
|
||||
_DEFAULT_TORCHVISION_VERSION = '0.24.0'
|
||||
_DEFAULT_TORCHAUDIO_VERSION = '2.9.0'
|
||||
_DEFAULT_TORCH_NPU_VERSION = '2.9.0.post2'
|
||||
_DEFAULT_VLLM_VERSION = '0.18.0'
|
||||
_DEFAULT_VLLM_ASCEND_VERSION = '0.18.0'
|
||||
_DEFAULT_TRITON_ASCEND_VERSIONS = {
|
||||
'8.5': '3.2.0',
|
||||
'9.0': '3.2.1',
|
||||
}
|
||||
_CANN_VERSION_PATTERN = re.compile(r'^\d+(?:\.[0-9A-Za-z]+)+$')
|
||||
_OS_TAG_PATTERN = re.compile(r'^[A-Za-z]+[0-9][0-9A-Za-z.]*$')
|
||||
_PYTHON_TAG_PATTERN = re.compile(r'^py\d+\.\d+$', re.IGNORECASE)
|
||||
_TORCH_NPU_VERSION_PATTERN = re.compile(
|
||||
r'^(?P<torch_version>\d+\.\d+\.\d+)(?:\.post\d+)?$')
|
||||
|
||||
@staticmethod
|
||||
def _normalize_arch(arch: str = None) -> str:
|
||||
arch = arch or platform.machine()
|
||||
arch = arch.lower()
|
||||
arch_mapping = {
|
||||
'x86': 'x86',
|
||||
'x86_64': 'x86',
|
||||
'amd64': 'x86',
|
||||
'arm': 'arm',
|
||||
'aarch64': 'arm',
|
||||
'arm64': 'arm',
|
||||
'x86': 'x86_64',
|
||||
'x86_64': 'x86_64',
|
||||
'amd64': 'x86_64',
|
||||
'arm': 'aarch64',
|
||||
'aarch64': 'aarch64',
|
||||
'arm64': 'aarch64',
|
||||
}
|
||||
if arch not in arch_mapping:
|
||||
raise ValueError(f'Unsupported architecture: {arch}. '
|
||||
@@ -536,34 +928,130 @@ class AscendImageBuilder(StableGPUImageBuilder):
|
||||
|
||||
cann_version = parts[0]
|
||||
os_tag = parts[2]
|
||||
python_tag = parts[3]
|
||||
if not cls._CANN_VERSION_PATTERN.fullmatch(cann_version):
|
||||
raise ValueError(f'Invalid CANN version in Ascend base image tag: '
|
||||
f'{cann_version}')
|
||||
if not cls._OS_TAG_PATTERN.fullmatch(os_tag):
|
||||
raise ValueError(
|
||||
f'Invalid OS tag in Ascend base image tag: {os_tag}')
|
||||
if not cls._PYTHON_TAG_PATTERN.fullmatch(python_tag):
|
||||
raise ValueError(
|
||||
f'Invalid Python tag in Ascend base image tag: {python_tag}')
|
||||
|
||||
return cann_version, f'CANN{cann_version}', os_tag
|
||||
return cann_version, f'CANN{cann_version}', os_tag, python_tag
|
||||
|
||||
@staticmethod
|
||||
def _get_os_family(os_tag: str) -> str:
|
||||
os_tag = os_tag.lower()
|
||||
if os_tag.startswith('ubuntu'):
|
||||
return 'ubuntu'
|
||||
if os_tag.startswith('openeuler'):
|
||||
return 'openeuler'
|
||||
raise ValueError(f'Unsupported Ascend base image OS tag: {os_tag}. '
|
||||
'Supported OS families are Ubuntu and openEuler.')
|
||||
|
||||
@classmethod
|
||||
def _init_torch_versions(cls, args) -> None:
|
||||
torch_version_specified = args.torch_version is not None
|
||||
torchvision_version_specified = args.torchvision_version is not None
|
||||
torchaudio_version_specified = args.torchaudio_version is not None
|
||||
|
||||
if torch_version_specified:
|
||||
if (not torchvision_version_specified
|
||||
or not torchaudio_version_specified):
|
||||
raise ValueError(
|
||||
'When overriding --torch_version for an Ascend image, also '
|
||||
'pass matching --torchvision_version and '
|
||||
'--torchaudio_version.')
|
||||
elif torchvision_version_specified or torchaudio_version_specified:
|
||||
raise ValueError(
|
||||
'--torchvision_version and --torchaudio_version require an '
|
||||
'explicit --torch_version for an Ascend image.')
|
||||
|
||||
args.torch_version = args.torch_version or cls._DEFAULT_TORCH_VERSION
|
||||
args.torchvision_version = (
|
||||
args.torchvision_version or cls._DEFAULT_TORCHVISION_VERSION)
|
||||
args.torchaudio_version = (
|
||||
args.torchaudio_version or cls._DEFAULT_TORCHAUDIO_VERSION)
|
||||
args.torch_npu_version = (
|
||||
args.torch_npu_version or cls._DEFAULT_TORCH_NPU_VERSION)
|
||||
|
||||
match = cls._TORCH_NPU_VERSION_PATTERN.fullmatch(
|
||||
args.torch_npu_version)
|
||||
if not match:
|
||||
raise ValueError('Invalid --torch_npu_version. Expected '
|
||||
'<major>.<minor>.<patch> or '
|
||||
'<major>.<minor>.<patch>.post<revision>.')
|
||||
if args.torch_version != match.group('torch_version'):
|
||||
raise ValueError(
|
||||
'--torch_version must exactly match the base version of '
|
||||
f'--torch_npu_version, got torch={args.torch_version} and '
|
||||
f'torch_npu={args.torch_npu_version}.')
|
||||
|
||||
@classmethod
|
||||
def _init_component_versions(cls, args) -> None:
|
||||
args.vllm_version = args.vllm_version or cls._DEFAULT_VLLM_VERSION
|
||||
args.vllm_ascend_version = (
|
||||
args.vllm_ascend_version or cls._DEFAULT_VLLM_ASCEND_VERSION)
|
||||
args.vllm_git_ref = cls._get_vllm_git_ref(args.vllm_version)
|
||||
args.vllm_ascend_git_ref = cls._get_vllm_git_ref(
|
||||
args.vllm_ascend_version)
|
||||
|
||||
if not args.triton_ascend_version:
|
||||
cann_series = '.'.join(args.cann_version.split('.')[:2])
|
||||
try:
|
||||
args.triton_ascend_version = (
|
||||
cls._DEFAULT_TRITON_ASCEND_VERSIONS[cann_series])
|
||||
except KeyError as e:
|
||||
raise ValueError('No default triton-ascend version for CANN '
|
||||
f'{args.cann_version}. Please pass '
|
||||
'--triton_ascend_version explicitly.') from e
|
||||
|
||||
@staticmethod
|
||||
def _get_vllm_git_ref(version: str) -> str:
|
||||
return version if version.startswith('v') else f'v{version}'
|
||||
|
||||
def init_args(self, args) -> Any:
|
||||
if not args.base_image:
|
||||
# Reuse the prebuilt vllm-ascend image to avoid rebuilding its stack.
|
||||
args.base_image = 'quay.io/ascend/cann:8.5.1-a3-ubuntu22.04-py3.11'
|
||||
self._init_torch_versions(args)
|
||||
args.arch = self._normalize_arch(args.arch)
|
||||
args.atlas_hardware = self._get_atlas_hardware(args.soc_version)
|
||||
args.cann_version, args.cann_version_tag, args.os_tag = (
|
||||
self._get_cann_os_tags(args.base_image))
|
||||
(args.cann_version, args.cann_version_tag, args.os_tag,
|
||||
args.ascend_python_tag) = (
|
||||
self._get_cann_os_tags(args.base_image))
|
||||
self._get_os_family(args.os_tag)
|
||||
self._init_component_versions(args)
|
||||
return super().init_args(args)
|
||||
|
||||
def _generate_python_tag(self, _python_version: str) -> str:
|
||||
return self.args.ascend_python_tag
|
||||
|
||||
def generate_dockerfile(self) -> str:
|
||||
extra_content = """
|
||||
RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
RUN export PIP_EXTRA_INDEX_URL=https://pypi.org/simple && \
|
||||
pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
"""
|
||||
with open('docker/Dockerfile.ascend', 'r') as f:
|
||||
content = f.read()
|
||||
content = content.replace('{base_image}', self.args.base_image)
|
||||
content = content.replace('{soc_version}', self.args.soc_version)
|
||||
content = content.replace('{cann_version}', self.args.cann_version)
|
||||
content = content.replace('{torch_version}',
|
||||
self.args.torch_version)
|
||||
content = content.replace('{torchvision_version}',
|
||||
self.args.torchvision_version)
|
||||
content = content.replace('{torchaudio_version}',
|
||||
self.args.torchaudio_version)
|
||||
content = content.replace('{torch_npu_version}',
|
||||
self.args.torch_npu_version)
|
||||
content = content.replace('{vllm_git_ref}', self.args.vllm_git_ref)
|
||||
content = content.replace('{vllm_ascend_git_ref}',
|
||||
self.args.vllm_ascend_git_ref)
|
||||
content = content.replace('{triton_ascend_version}',
|
||||
self.args.triton_ascend_version)
|
||||
content = content.replace('{extra_content}', extra_content)
|
||||
content = content.replace('{cur_time}', formatted_time)
|
||||
content = content.replace('{install_ms_deps}', 'False')
|
||||
@@ -577,11 +1065,12 @@ RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
|
||||
return content
|
||||
|
||||
def image(self) -> str:
|
||||
return (
|
||||
f'{docker_registry}:{self.args.swift_branch}-'
|
||||
f'{self.args.atlas_hardware}-{self.args.python_tag}-'
|
||||
f'{self.args.cann_version_tag}-{self.args.os_tag}-{self.args.arch}'
|
||||
)
|
||||
tag = (f'{self.args.swift_branch}-{self.args.cann_version_tag}-'
|
||||
f'torch_npu{self.args.torch_npu_version}-'
|
||||
f'{self.args.atlas_hardware}-{self.args.os_tag}-'
|
||||
f'{self.args.python_tag}-'
|
||||
f'{self.args.arch}')
|
||||
return f'{docker_registry}:{tag.lower()}'
|
||||
|
||||
def push(self):
|
||||
return 0
|
||||
@@ -593,6 +1082,7 @@ parser.add_argument('--image_type', type=str)
|
||||
parser.add_argument('--python_version', type=str, default='3.12.13')
|
||||
parser.add_argument('--ubuntu_version', type=str, default='22.04')
|
||||
parser.add_argument('--torch_version', type=str, default=None)
|
||||
parser.add_argument('--torch_npu_version', type=str, default=None)
|
||||
parser.add_argument('--torchvision_version', type=str, default=None)
|
||||
parser.add_argument('--cuda_version', type=str, default=None)
|
||||
parser.add_argument('--ci_image', type=int, default=0)
|
||||
@@ -600,6 +1090,8 @@ parser.add_argument('--torchaudio_version', type=str, default=None)
|
||||
parser.add_argument('--optimum_version', type=str, default=None)
|
||||
parser.add_argument('--tf_version', type=str, default=None)
|
||||
parser.add_argument('--vllm_version', type=str, default=None)
|
||||
parser.add_argument('--vllm_ascend_version', type=str, default=None)
|
||||
parser.add_argument('--triton_ascend_version', type=str, default=None)
|
||||
parser.add_argument('--lmdeploy_version', type=str, default=None)
|
||||
parser.add_argument('--flashattn_version', type=str, default=None)
|
||||
parser.add_argument('--autogptq_version', type=str, default=None)
|
||||
@@ -610,6 +1102,12 @@ parser.add_argument('--megatron_branch', type=str, default='v0.15.3')
|
||||
parser.add_argument('--mindspeed_branch', type=str, default='core_r0.15.3')
|
||||
parser.add_argument('--soc_version', type=str, default='ascend910_9391')
|
||||
parser.add_argument('--arch', type=str, choices=['x86', 'arm'], default=None)
|
||||
parser.add_argument(
|
||||
'--base_image_tag',
|
||||
type=str,
|
||||
default=None,
|
||||
help='Optional AMD ROCm override tag. Default: auto-resolve newest '
|
||||
'concrete vllm/vllm-openai-rocm release tag from Docker Hub.')
|
||||
parser.add_argument('--dry_run', type=int, default=0)
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -621,6 +1119,8 @@ elif args.image_type.lower() == 'stable':
|
||||
builder_cls = [StableCPUImageBuilder, StableGPUImageBuilder]
|
||||
elif args.image_type.lower() == 'ascend':
|
||||
builder_cls = [AscendImageBuilder]
|
||||
elif args.image_type.lower() == 'amd':
|
||||
builder_cls = [AmdImageBuilder]
|
||||
elif args.image_type.lower() == 'latest':
|
||||
builder_cls = [LatestGPUImageBuilder]
|
||||
else:
|
||||
|
||||
@@ -47,5 +47,4 @@ else
|
||||
fi
|
||||
|
||||
pip config set global.index-url https://mirrors.cloud.aliyuncs.com/pypi/simple
|
||||
pip config set global.extra-index-url https://pypi.org/simple
|
||||
pip config set install.trusted-host mirrors.cloud.aliyuncs.com
|
||||
|
||||
Reference in New Issue
Block a user