merge master

This commit is contained in:
hjh0119
2026-08-03 20:49:52 +08:00
24 changed files with 1804 additions and 225 deletions

26
docker/Dockerfile.amd Normal file
View File

@@ -0,0 +1,26 @@
FROM {base_image}
ARG BASE_IMAGE_TAG={base_image_tag}
LABEL modelscope.base_image="vllm/vllm-openai-rocm:${BASE_IMAGE_TAG}"
# Build-time only (ARG does not persist into the image). Aliyun mirror can lag PyPI (~1h).
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
COPY docker/scripts/modelscope_env_init.sh /usr/local/bin/ms_env_init.sh
ARG CUR_TIME={cur_time}
RUN echo "CUR_TIME=${CUR_TIME}" && echo "BASE_IMAGE_TAG=${BASE_IMAGE_TAG}"
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
pip config set global.index-url https://mirrors.aliyun.com/pypi/simple && \
pip config set install.trusted-host mirrors.aliyun.com && \
cd /tmp && GIT_LFS_SKIP_SMUDGE=1 git clone -b {modelscope_branch} --single-branch https://github.com/modelscope/modelscope.git && \
cd modelscope && pip install --no-cache-dir . -f https://modelscope.oss-cn-beijing.aliyuncs.com/releases/repo.html && \
cd / && rm -fr /tmp/modelscope && pip cache purge
ENV VLLM_USE_MODELSCOPE=True
ENV LMDEPLOY_USE_MODELSCOPE=True
ENV MODELSCOPE_CACHE=/mnt/workspace/.cache/modelscope/hub
SHELL ["/bin/bash", "-c"]

View File

@@ -5,39 +5,54 @@ ENV PIP_DISABLE_PIP_VERSION_CHECK=1 \
PIP_RETRIES=10 \
SOC_VERSION={soc_version} \
CANN_VERSION={cann_version}
# Build-time only (ARG does not persist into the image).
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
SHELL ["/bin/bash", "-c"]
# ---------- System dependencies ----------
RUN rm -f /etc/apt/apt.conf.d/docker-clean && \
find /etc/apt/apt.conf.d -maxdepth 1 -type f | xargs -r grep -l "APT::Update::Post-Invoke\|docker-clean" | xargs -r rm -f && \
apt-get update -y && \
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
gcc g++ cmake ninja-build libnuma-dev libgl1 libglib2.0-0 libsm6 libxext6 libxrender1 \
wget git curl jq vim build-essential ca-certificates && \
apt-get clean && \
rm -rf /var/lib/apt/lists/*
RUN set -eux; \
. /etc/os-release; \
case "${ID,,}" in \
ubuntu) \
rm -f /etc/apt/apt.conf.d/docker-clean; \
find /etc/apt/apt.conf.d -maxdepth 1 -type f | xargs -r grep -l "APT::Update::Post-Invoke\|docker-clean" | xargs -r rm -f; \
apt-get update -y; \
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
gcc g++ cmake ninja-build libnuma-dev libgl1 libglib2.0-0 libsm6 libxext6 libxrender1 \
wget git curl jq vim build-essential ca-certificates; \
apt-get clean; \
rm -rf /var/lib/apt/lists/* \
;; \
openeuler) \
yum install -y \
gcc gcc-c++ cmake ninja-build numactl-devel mesa-libGL glib2 libSM libXext libXrender \
wget git curl jq vim make ca-certificates; \
yum clean all; \
rm -rf /var/cache/yum \
;; \
*) \
echo "Unsupported base image OS: ${ID}" >&2; \
exit 1 \
;; \
esac
RUN pip config set global.index-url https://mirrors.aliyun.com/pypi/simple && \
pip config set global.extra-index-url "https://pypi.org/simple" && \
pip config set install.trusted-host mirrors.aliyun.com && \
ARCH=$(uname -m) && \
if [ "$ARCH" = "x86_64" ]; then \
pip config set global.extra-index-url "https://pypi.org/simple https://download.pytorch.org/whl/cpu/"; \
fi
pip config set install.trusted-host mirrors.aliyun.com
{extra_content}
# ---------- Install vllm + vllm-ascend ----------
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
git clone --depth 1 --branch v0.18.0 https://github.com/vllm-project/vllm && \
git clone --depth 1 --branch v0.18.0 https://github.com/vllm-project/vllm-ascend.git
git clone --depth 1 --branch {vllm_git_ref} https://github.com/vllm-project/vllm && \
git clone --depth 1 --branch {vllm_ascend_git_ref} https://github.com/vllm-project/vllm-ascend.git
RUN ARCH=$(uname -m) && \
export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
source /usr/local/Ascend/nnal/atb/set_env.sh && \
# Install torch & torch_npu & torchvision
pip install torch==2.9.0 torch_npu==2.9.0.post2 torchvision==0.24.0 && \
pip install torch=={torch_version} torch_npu=={torch_npu_version} torchvision=={torchvision_version} && \
# Install vllm
cd vllm && VLLM_TARGET_DEVICE=empty pip install -v -e . && cd .. && \
# Install vllm-ascend
@@ -46,32 +61,33 @@ RUN ARCH=$(uname -m) && \
# ---------- Clone training-side repositories ----------
RUN git clone --depth 1 --branch {megatron_branch} https://github.com/NVIDIA/Megatron-LM.git /Megatron-LM && \
git clone --depth 1 --branch {mindspeed_branch} https://gitcode.com/Ascend/MindSpeed.git /MindSpeed && \
GIT_LFS_SKIP_SMUDGE=1 git clone --depth 1 -b {swift_branch} --single-branch https://github.com/modelscope/ms-swift.git /ms-swift && \
git clone --depth 1 https://github.com/modelscope/mcore-bridge.git /mcore-bridge
GIT_LFS_SKIP_SMUDGE=1 git clone --depth 1 -b {swift_branch} --single-branch https://github.com/modelscope/ms-swift.git /ms-swift
# ---------- Install training-side repositories ----------
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
cd /MindSpeed && pip install --no-cache-dir -e . && \
cd /mcore-bridge && pip install --no-cache-dir -e . && \
pip install --no-cache-dir mcore-bridge -i https://pypi.org/simple/ -U && \
cd /ms-swift && pip install --no-cache-dir -e .
# ---------- Pin torch to the correct version + torch_npu ----------
# x86: must force-install the CPU build from pytorch.org/whl/cpu
# aarch64: PyPI only provides the CPU build, so install it directly from the Aliyun mirror
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
ARCH=$(uname -m) && \
if [ "$ARCH" = "x86_64" ]; then \
pip install --no-cache-dir --force-reinstall --no-deps \
--index-url https://download.pytorch.org/whl/cpu \
torch==2.9.0 torchvision==0.24.0 torchaudio==2.9.0; \
torch=={torch_version} torchvision=={torchvision_version} torchaudio=={torchaudio_version}; \
else \
pip install --no-cache-dir --force-reinstall --no-deps \
torch==2.9.0 torchvision==0.24.0 torchaudio==2.9.0; \
torch=={torch_version} torchvision=={torchvision_version} torchaudio=={torchaudio_version}; \
fi && \
pip install --no-cache-dir --force-reinstall --no-deps \
torch_npu==2.9.0.post2 && \
torch_npu=={torch_npu_version} && \
rm -rf /root/.cache/pip
# ---------- Remove CUDA-only dependencies pulled in by vllm (they cause missing libtorch_cuda.so errors on NPU) ----------
@@ -83,7 +99,8 @@ ENV PYTHONPATH=/Megatron-LM:${PYTHONPATH}
# install dependencies
COPY requirements /var/modelscope
RUN pip uninstall ms-swift modelscope -y && pip install --no-cache-dir pip==23.* -U && \
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
pip uninstall ms-swift modelscope -y && pip install --no-cache-dir pip==23.* -U && \
if [ "$INSTALL_MS_DEPS" = "True" ]; then \
pip install --no-cache-dir omegaconf==2.0.6 && \
pip install 'editdistance==0.8.1' && \
@@ -109,9 +126,11 @@ fi
ARG CUR_TIME={cur_time}
RUN echo $CUR_TIME
RUN pip install --no-cache-dir --no-build-isolation OpenCC
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
pip install --no-cache-dir --no-build-isolation OpenCC
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
pip install --no-cache-dir -U funasr scikit-learn && \
pip install --no-cache-dir -U qwen_vl_utils qwen_omni_utils librosa 'timm>=0.9.0' transformers accelerate peft trl safetensors && \
@@ -126,36 +145,20 @@ RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
pip install --no-cache-dir omegaconf==2.3.0 && \
pip cache purge
# ---------- Reinstall triton-ascend for the selected CANN version ----------
# ---------- Install training and evaluation dependencies ----------
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \
if [ -f /usr/local/Ascend/nnal/atb/set_env.sh ]; then source /usr/local/Ascend/nnal/atb/set_env.sh; fi && \
TORCH_DEVICE_BACKEND_AUTOLOAD=0 pip install --no-cache-dir "deepspeed<0.19" ray liger_kernel pre-commit -U && \
pip cache purge
# ---------- Install triton-ascend ----------
RUN set -eux; \
export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}"; \
pip uninstall -y triton || true; \
pip uninstall -y triton-ascend || true; \
case "${CANN_VERSION}" in \
8.5.*) \
pip install --no-cache-dir --force-reinstall triton-ascend==3.2.0; \
;; \
9.0.0) \
PY_ABI="cp$(python -c 'import sys; print(f"{sys.version_info.major}{sys.version_info.minor}")')"; \
case "${PY_ABI}" in \
cp310|cp311|cp312|cp313) ;; \
*) echo "Unsupported Python ABI for triton-ascend 3.2.1: ${PY_ABI}" >&2; exit 1 ;; \
esac; \
ARCH="$(uname -m)"; \
case "${ARCH}" in \
aarch64|x86_64) ;; \
*) echo "Unsupported architecture for triton-ascend 3.2.1: ${ARCH}" >&2; exit 1 ;; \
esac; \
WHEEL_NAME="triton_ascend-3.2.1-${PY_ABI}-${PY_ABI}-manylinux_2_27_${ARCH}.manylinux_2_28_${ARCH}.whl"; \
WHEEL_PATH="/tmp/${WHEEL_NAME}"; \
curl -fL "https://gitcode.com/Ascend/triton-ascend/releases/download/v3.2.1/${WHEEL_NAME}" -o "${WHEEL_PATH}"; \
pip install --no-cache-dir --force-reinstall "${WHEEL_PATH}"; \
rm -f "${WHEEL_PATH}"; \
;; \
*) \
echo "Unsupported CANN_VERSION for triton-ascend install: ${CANN_VERSION}" >&2; \
exit 1; \
;; \
esac
pip install --no-cache-dir --force-reinstall \
triton-ascend=={triton_ascend_version} \
--extra-index-url=https://triton-ascend.osinfra.cn/pypi/simple
RUN echo 'source /usr/local/Ascend/ascend-toolkit/set_env.sh' >> /root/.bashrc && \
echo '[ -f /usr/local/Ascend/nnal/atb/set_env.sh ] && source /usr/local/Ascend/nnal/atb/set_env.sh' >> /root/.bashrc && \

View File

@@ -3,6 +3,8 @@ FROM {base_image}
ARG DEBIAN_FRONTEND=noninteractive
ENV TZ=Asia/Shanghai
ENV arch=x86_64
# Build-time only (ARG does not persist into the image). Aliyun mirror can lag PyPI (~1h).
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
COPY docker/scripts/modelscope_env_init.sh /usr/local/bin/ms_env_init.sh
RUN apt-get update && \
@@ -21,7 +23,9 @@ ARG IMAGE_TYPE={image_type}
# install dependencies
COPY requirements /var/modelscope
RUN pip uninstall ms-swift modelscope -y && pip --no-cache-dir install pip==23.* -U && \
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
pip uninstall ms-swift modelscope -y && pip --no-cache-dir install pip==23.* -U && \
if [ "$INSTALL_MS_DEPS" = "True" ]; then \
pip --no-cache-dir install omegaconf==2.0.6 && \
pip install 'editdistance==0.8.1' && \
@@ -34,7 +38,7 @@ if [ "$INSTALL_MS_DEPS" = "True" ]; then \
pip install --no-cache-dir 'scipy' && \
pip install --no-cache-dir funtextprocessing typeguard==2.13.3 scikit-learn -f https://modelscope.oss-cn-beijing.aliyuncs.com/releases/repo.html && \
pip install --no-cache-dir 'decord>=0.6.0' mpi4py paint_ldm ipykernel fasttext -f https://modelscope.oss-cn-beijing.aliyuncs.com/releases/repo.html && \
pip install --no-cache-dir ipywidgets && \
pip install --no-cache-dir ipywidgets jupyter_core nbconvert nbclient && \
pip install --no-cache-dir 'blobfile>=1.0.5' && \
pip uninstall MinDAEC -y && \
pip install https://modelscope.oss-cn-beijing.aliyuncs.com/releases/dependencies/MinDAEC-0.0.2-py3-none-any.whl && \
@@ -47,7 +51,9 @@ fi
ARG CUR_TIME={cur_time}
RUN echo $CUR_TIME
RUN bash /tmp/install.sh {version_args} && \
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
bash /tmp/install.sh {version_args} && \
pip install --no-cache-dir -U funasr scikit-learn && \
pip install --no-cache-dir -U qwen_vl_utils qwen_omni_utils librosa timm transformers accelerate peft trl safetensors && \
cd /tmp && GIT_LFS_SKIP_SMUDGE=1 git clone -b {swift_branch} --single-branch https://github.com/modelscope/ms-swift.git && \
@@ -63,23 +69,17 @@ RUN bash /tmp/install.sh {version_args} && \
pip install --no-cache-dir transformers diffusers 'timm>=0.9.0' && pip cache purge; \
pip install --no-cache-dir omegaconf==2.3.0 && pip cache purge; \
pip config set global.index-url https://mirrors.aliyun.com/pypi/simple && \
pip config set global.extra-index-url https://pypi.org/simple && \
pip config set install.trusted-host mirrors.aliyun.com && \
cp /tmp/resources/ubuntu2204.aliyun /etc/apt/sources.list
RUN if [ "$IMAGE_TYPE" = "gpu" ]; then \
pip install --no-cache-dir math_verify "datasets<4.8.5" "gradio<5.33" "deepspeed<0.19" ray -U && \
ARG PIP_EXTRA_INDEX_URL=https://pypi.org/simple
RUN export PIP_EXTRA_INDEX_URL="${PIP_EXTRA_INDEX_URL}" && \
if [ "$IMAGE_TYPE" = "gpu" ]; then \
pip install --no-cache-dir math_verify "gradio<5.33" "deepspeed<0.19" ray -U && \
pip install --no-cache-dir mcore-bridge -i https://pypi.org/simple/ -U && \
pip install --no-cache-dir pybind11 liger_kernel wandb swanlab nvitop pre-commit "transformers<5.15" "trl<1.0" "peft<0.21" huggingface-hub -U && \
pip install git+https://github.com/NVIDIA/TransformerEngine.git@stable --no-build-isolation; \
pip install git+https://github.com/deepseek-ai/DeepGEMM.git@v2.1.1.post3 --no-build-isolation; \
pip install -U flash-linear-attention --no-build-isolation; \
pip install -U git+https://github.com/Dao-AILab/causal-conv1d --no-build-isolation; \
pip install git+https://github.com/Dao-AILab/fast-hadamard-transform --no-build-isolation; \
pip install git+https://github.com/NVIDIA-NeMo/Emerging-Optimizers.git@v0.3.0; \
mv /usr/local/lib/python3.12/site-packages/tilelang/lib/libcudart_stub.so /usr/local/lib/python3.12/site-packages/tilelang/lib/libcudart_stub.so.bak; \
ln -s /usr/local/cuda-13.0/targets/x86_64-linux/lib/libcudart.so /usr/local/lib/python3.12/site-packages/tilelang/lib/libcudart_stub.so; \
pip install --no-cache-dir liger_kernel wandb swanlab nvitop pre-commit "transformers" "trl<1.0" "peft<0.21" huggingface-hub -U && \
pip install --no-cache-dir --no-build-isolation "transformer_engine[pytorch]==2.16.0"; \
cd /tmp && GIT_LFS_SKIP_SMUDGE=1 git clone https://github.com/NVIDIA/apex && \
cd apex && pip install -v --disable-pip-version-check --no-cache-dir --no-build-isolation --config-settings "--build-option=--cpp_ext" --config-settings "--build-option=--cuda_ext" ./ && \
cd / && rm -fr /tmp/apex && pip cache purge; \

View File

@@ -10,7 +10,8 @@ ms-swift Ascend images provide a ready-to-use ms-swift environment for Huawei As
- Build template: `docker/Dockerfile.ascend`
- Build entrypoint: `docker/build_image.py --image_type ascend`
- Default base image: `quay.io/ascend/cann:8.5.1-a3-ubuntu22.04-py3.11`
- Default output tag: `${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-<arch>`
- Supported base OSes: Ubuntu and openEuler, selected from the CANN base-image tag
- Default output tag: `${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-<arch>`
- Ascend runtime environment is sourced from `/usr/local/Ascend/ascend-toolkit/set_env.sh`
- If available, NNAL/ATB runtime is sourced from `/usr/local/Ascend/nnal/atb/set_env.sh`
@@ -22,45 +23,46 @@ The Ascend Dockerfile installs and configures:
| --- | --- |
| CANN | inherited from the selected `quay.io/ascend/cann` base image |
| Python | inherited from the base image tag, for example `py3.11` |
| PyTorch | `torch==2.9.0` |
| torch-npu | `torch_npu==2.9.0.post2` |
| torchvision / torchaudio | `torchvision==0.24.0`, `torchaudio==2.9.0` |
| vLLM | source install from `vllm-project/vllm`, default branch `v0.18.0` |
| vLLM Ascend | source install from `vllm-project/vllm-ascend`, default branch `v0.18.0` |
| PyTorch | `torch==2.9.0` by default; configurable with `--torch_version` |
| torch-npu | `torch_npu==2.9.0.post2` by default; configurable with `--torch_npu_version` |
| torchvision / torchaudio | `torchvision==0.24.0`, `torchaudio==2.9.0` by default; pass both explicitly when overriding `--torch_version` |
| vLLM | source install from `vllm-project/vllm`, default `0.18.0`; configurable with `--vllm_version` |
| vLLM Ascend | source install from `vllm-project/vllm-ascend`, default `0.18.0`; configurable with `--vllm_ascend_version` |
| Megatron-LM | source checkout, default branch `v0.15.3` |
| MindSpeed | source checkout, default branch `core_r0.15.3` |
| mcore-bridge | source checkout from `modelscope/mcore-bridge` |
| mcore-bridge | latest release from PyPI |
| ms-swift | source checkout from `modelscope/ms-swift`, default branch `main` |
| ModelScope | source checkout from `modelscope/modelscope`, default branch `master` |
| triton-ascend | `3.2.0` for CANN `8.5.*`; local wheel install of `3.2.1` for CANN `9.0.0` |
| triton-ascend | CANN `8.5.*` defaults to `3.2.0`; CANN `9.0.*` defaults to `3.2.1`; configurable with `--triton_ascend_version` and installed from the Triton Ascend PyPI index |
## Supported Tag Format
Images built by `docker/build_image.py --image_type ascend` use this tag format:
```text
${DOCKER_REGISTRY}:<swift-branch>-<atlas-hardware>-<python-tag>-<cann-version-tag>-<os-tag>-<arch>
${DOCKER_REGISTRY}:<swift-branch>-<cann-version-tag>-torch_npu<torch-npu-version>-<atlas-hardware>-<os-tag>-<python-tag>-<arch>
```
| Field | Example | Description |
| --- | --- | --- |
| `swift-branch` | `main` | ms-swift branch used during image build |
| `atlas-hardware` | `A2`, `A3`, `300I`, `A5` | Derived from `--soc_version` |
| `python-tag` | `py311` | Derived from `--python_version` |
| `cann-version-tag` | `CANN8.5.1`, `CANN9.0.0` | Parsed from the CANN base image tag |
| `os-tag` | `ubuntu22.04` | Parsed from the CANN base image tag |
| `arch` | `arm`, `x86` | Derived from host architecture or `--arch` |
| `cann-version-tag` | `cann8.5.1`, `cann9.0.0` | Parsed from the CANN base image tag |
| `torch-npu-version` | `2.9.0.post2` | From `--torch_npu_version`; defaults to `2.9.0.post2` |
| `atlas-hardware` | `a2`, `a3`, `300i`, `a5` | Derived from `--soc_version` |
| `os-tag` | `ubuntu22.04`, `openeuler24.03` | Parsed from the CANN base-image tag; prevents tags for different OSes from colliding |
| `python-tag` | `py3.11` | Parsed from the CANN base image tag |
| `arch` | `aarch64`, `x86_64` | Derived from host architecture or `--arch` |
Default example on an ARM64 host:
```text
${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-arm
${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-aarch64
```
A2 / CANN 9.0.0 example:
```text
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64
```
## Build Locally
@@ -85,6 +87,39 @@ python docker/build_image.py \
--soc_version ascend910b1
```
Build an openEuler image. The system-dependency layer automatically uses `yum`; Ubuntu images continue to use `apt-get`.
```bash
python docker/build_image.py \
--image_type ascend \
--base_image quay.io/ascend/cann:8.5.1-a3-openeuler24.03-py3.11 \
--soc_version ascend910_9391
```
Override the PyTorch stack. `--torch_version` must match the base version of
`--torch_npu_version`; when overriding PyTorch, pass its matching torchvision
and torchaudio versions explicitly.
```bash
python docker/build_image.py \
--image_type ascend \
--torch_version 2.9.0 \
--torch_npu_version 2.9.0.post2 \
--torchvision_version 0.24.0 \
--torchaudio_version 2.9.0
```
Override the vLLM stack or triton-ascend. The vLLM version arguments select
the matching Git tag, for example `0.18.0` selects `v0.18.0`.
```bash
python docker/build_image.py \
--image_type ascend \
--vllm_version 0.18.0 \
--vllm_ascend_version 0.18.0 \
--triton_ascend_version 3.2.1
```
Override Megatron or MindSpeed source branches when needed:
```bash
@@ -94,11 +129,11 @@ python docker/build_image.py \
--mindspeed_branch core_r0.15.3
```
For slow networks, Linux hosts can use Docker host networking after the root `Dockerfile` is generated:
To run the rendered Dockerfile manually, use:
```bash
docker build --network host \
-t ${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
docker build \
-t ${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
-f Dockerfile .
```
@@ -119,7 +154,7 @@ docker run --rm -it \
-v /usr/local/Ascend/driver/version.info:/usr/local/Ascend/driver/version.info \
-v /etc/ascend_install.info:/etc/ascend_install.info \
-v /mnt/workspace:/mnt/workspace \
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
bash
```
@@ -147,7 +182,8 @@ pip show ms-swift modelscope torch-npu triton-ascend
## Notes
- CANN, firmware, and driver versions must be compatible with each other.
- CANN `8.5.*` and CANN `9.0.0` use different `triton-ascend` install paths in this Dockerfile.
- Ubuntu base images install system dependencies through `apt-get`; openEuler base images install the corresponding RPM packages through `yum`.
- `triton-ascend` is installed from `https://triton-ascend.osinfra.cn/pypi/simple`; select a version compatible with the chosen CANN, Python, and architecture.
- The image is intended for Ascend NPU ms-swift workflows. CUDA-only packages pulled in by dependencies are removed when they conflict with NPU runtime libraries.
- Use a fixed image tag for production jobs instead of relying on a moving branch name.

View File

@@ -10,7 +10,8 @@ ms-swift Ascend 镜像面向华为昇腾 Atlas NPU提供可直接使用的 ms
- 构建模板:`docker/Dockerfile.ascend`
- 构建入口:`docker/build_image.py --image_type ascend`
- 默认基础镜像:`quay.io/ascend/cann:8.5.1-a3-ubuntu22.04-py3.11`
- 默认输出 tag`${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-<arch>`
- 支持的基础 OSUbuntu 和 openEuler由 CANN 基础镜像 tag 选择
- 默认输出 tag`${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-<arch>`
- Ascend runtime 环境来自 `/usr/local/Ascend/ascend-toolkit/set_env.sh`
- 如果镜像内存在 NNAL/ATB则会加载 `/usr/local/Ascend/nnal/atb/set_env.sh`
@@ -22,45 +23,46 @@ Ascend Dockerfile 会安装和配置:
| --- | --- |
| CANN | 继承自选定的 `quay.io/ascend/cann` 基础镜像 |
| Python | 继承自基础镜像 tag例如 `py3.11` |
| PyTorch | `torch==2.9.0` |
| torch-npu | `torch_npu==2.9.0.post2` |
| torchvision / torchaudio | `torchvision==0.24.0``torchaudio==2.9.0` |
| vLLM | 从 `vllm-project/vllm` 源码安装,默认分支 `v0.18.0` |
| vLLM Ascend | 从 `vllm-project/vllm-ascend` 源码安装,默认分支 `v0.18.0` |
| PyTorch | 默认 `torch==2.9.0`;可通过 `--torch_version` 配置 |
| torch-npu | 默认 `torch_npu==2.9.0.post2`;可通过 `--torch_npu_version` 配置 |
| torchvision / torchaudio | 默认 `torchvision==0.24.0``torchaudio==2.9.0`;覆盖 `--torch_version` 时必须同时显式传入两者 |
| vLLM | 从 `vllm-project/vllm` 源码安装,默认 `0.18.0`;可通过 `--vllm_version` 配置 |
| vLLM Ascend | 从 `vllm-project/vllm-ascend` 源码安装,默认 `0.18.0`;可通过 `--vllm_ascend_version` 配置 |
| Megatron-LM | 源码 checkout默认分支 `v0.15.3` |
| MindSpeed | 源码 checkout默认分支 `core_r0.15.3` |
| mcore-bridge | 来自 `modelscope/mcore-bridge` 的源码 checkout |
| mcore-bridge | PyPI 上的最新发布版 |
| ms-swift | 来自 `modelscope/ms-swift` 的源码 checkout默认分支 `main` |
| ModelScope | 来自 `modelscope/modelscope` 的源码 checkout默认分支 `master` |
| triton-ascend | CANN `8.5.*` 安装 `3.2.0`CANN `9.0.0` 下载并本地安装 `3.2.1` wheel |
| triton-ascend | CANN `8.5.*` 默认 `3.2.0`CANN `9.0.*` 默认 `3.2.1`;可通过 `--triton_ascend_version` 配置,并从 Triton Ascend PyPI 源安装 |
## 支持的 Tag 格式
通过 `docker/build_image.py --image_type ascend` 构建的镜像使用以下 tag 格式:
```text
${DOCKER_REGISTRY}:<swift-branch>-<atlas-hardware>-<python-tag>-<cann-version-tag>-<os-tag>-<arch>
${DOCKER_REGISTRY}:<swift-branch>-<cann-version-tag>-torch_npu<torch-npu-version>-<atlas-hardware>-<os-tag>-<python-tag>-<arch>
```
| 字段 | 示例 | 说明 |
| --- | --- | --- |
| `swift-branch` | `main` | 构建镜像时使用的 ms-swift 分支 |
| `atlas-hardware` | `A2``A3``300I``A5` | 从 `--soc_version` 推导 |
| `python-tag` | `py311` | `--python_version` 推导 |
| `cann-version-tag` | `CANN8.5.1``CANN9.0.0` | 从 CANN 基础镜像 tag 解析 |
| `os-tag` | `ubuntu22.04` | 从 CANN 基础镜像 tag 解析 |
| `arch` | `arm``x86` | 从宿主机架构或 `--arch` 推导 |
| `cann-version-tag` | `cann8.5.1``cann9.0.0` | 从 CANN 基础镜像 tag 解析 |
| `torch-npu-version` | `2.9.0.post2` | 来自 `--torch_npu_version`,默认 `2.9.0.post2` |
| `atlas-hardware` | `a2``a3``300i``a5` | 从 `--soc_version` 推导 |
| `os-tag` | `ubuntu22.04``openeuler24.03` | 从 CANN 基础镜像 tag 解析;避免不同 OS 的镜像 tag 冲突 |
| `python-tag` | `py3.11` | 从 CANN 基础镜像 tag 解析 |
| `arch` | `aarch64``x86_64` | 从宿主机架构或 `--arch` 推导 |
ARM64 宿主机上的默认示例:
```text
${DOCKER_REGISTRY}:main-A3-py311-CANN8.5.1-ubuntu22.04-arm
${DOCKER_REGISTRY}:main-cann8.5.1-torch_npu2.9.0.post2-a3-ubuntu22.04-py3.11-aarch64
```
A2 / CANN 9.0.0 示例:
```text
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64
```
## 本地构建
@@ -85,6 +87,36 @@ python docker/build_image.py \
--soc_version ascend910b1
```
构建 openEuler 镜像。系统依赖层会自动使用 `yum`Ubuntu 镜像继续使用 `apt-get`
```bash
python docker/build_image.py \
--image_type ascend \
--base_image quay.io/ascend/cann:8.5.1-a3-openeuler24.03-py3.11 \
--soc_version ascend910_9391
```
覆盖 PyTorch 版本组。`--torch_version` 必须与 `--torch_npu_version` 的基础版本一致;覆盖 PyTorch 时,必须显式传入匹配的 torchvision 和 torchaudio 版本:
```bash
python docker/build_image.py \
--image_type ascend \
--torch_version 2.9.0 \
--torch_npu_version 2.9.0.post2 \
--torchvision_version 0.24.0 \
--torchaudio_version 2.9.0
```
覆盖 vLLM 版本组或 triton-ascend。vLLM 版本参数会选择对应 Git tag例如 `0.18.0` 会选择 `v0.18.0`
```bash
python docker/build_image.py \
--image_type ascend \
--vllm_version 0.18.0 \
--vllm_ascend_version 0.18.0 \
--triton_ascend_version 3.2.1
```
需要时可以覆盖 Megatron 或 MindSpeed 源码分支:
```bash
@@ -94,11 +126,11 @@ python docker/build_image.py \
--mindspeed_branch core_r0.15.3
```
果构建时网络较慢Linux 宿主机可以在根目录 `Dockerfile` 生成后使用 host network 构建
需手工构建生成后的根目录 `Dockerfile`,可使用
```bash
docker build --network host \
-t ${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
docker build \
-t ${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
-f Dockerfile .
```
@@ -119,7 +151,7 @@ docker run --rm -it \
-v /usr/local/Ascend/driver/version.info:/usr/local/Ascend/driver/version.info \
-v /etc/ascend_install.info:/etc/ascend_install.info \
-v /mnt/workspace:/mnt/workspace \
${DOCKER_REGISTRY}:main-A2-py311-CANN9.0.0-ubuntu22.04-arm \
${DOCKER_REGISTRY}:main-cann9.0.0-torch_npu2.9.0.post2-a2-ubuntu22.04-py3.11-aarch64 \
bash
```
@@ -147,7 +179,8 @@ pip show ms-swift modelscope torch-npu triton-ascend
## 注意事项
- CANN、firmware 和 driver 版本必须互相兼容。
- 这个 Dockerfile 对 CANN `8.5.*` 和 CANN `9.0.0` 使用不同的 `triton-ascend` 安装路径
- Ubuntu 基础镜像通过 `apt-get` 安装系统依赖openEuler 基础镜像通过 `yum` 安装对应 RPM 包
- `triton-ascend``https://triton-ascend.osinfra.cn/pypi/simple` 安装;请选择与 CANN、Python 和架构兼容的版本。
- 该镜像面向 Ascend NPU 上的 ms-swift 工作流。依赖安装过程中引入且与 NPU runtime 冲突的 CUDA-only 包会被移除。
- 生产任务建议使用固定镜像 tag不要依赖浮动分支名。

View File

@@ -3,14 +3,27 @@ import os
import platform
import re
import subprocess
import urllib.error
import urllib.request
from copy import copy
from datetime import datetime
from typing import Any
from typing import Any, List, Optional
import json
docker_registry = os.environ['DOCKER_REGISTRY']
assert docker_registry, 'You must pass a valid DOCKER_REGISTRY'
timestamp = datetime.now()
formatted_time = timestamp.strftime('%Y%m%d%H%M%S')
VLLM_ROCM_REPO = 'vllm/vllm-openai-rocm'
_FLOATING_ROCM_TAGS = frozenset({
'latest',
'latest-base',
'nightly',
'base-nightly',
})
_VERSION_TAG_PATTERN = re.compile(r'^v\d+(?:\.\d+)*$')
_NIGHTLY_HASH_PATTERN = re.compile(r'^(?:base-)?nightly-[0-9a-f]{7,40}$')
class Builder:
@@ -359,7 +372,8 @@ class StableGPUImageBuilder(Builder):
extra_content = extra_content.replace('{python_version}',
self.args.python_version)
extra_content += """
RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
RUN export PIP_EXTRA_INDEX_URL=https://pypi.org/simple && \
pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
"""
version_args = (
f'{self.args.torch_version} {self.args.torchvision_version} {self.args.torchaudio_version} '
@@ -434,7 +448,8 @@ class LatestGPUImageBuilder(StableGPUImageBuilder):
extra_content = extra_content.replace('{python_version}',
self.args.python_version)
extra_content += """
RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
RUN export PIP_EXTRA_INDEX_URL=https://pypi.org/simple && \
pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
"""
version_args = (
f'{self.args.torch_version} {self.args.torchvision_version} {self.args.torchaudio_version} '
@@ -481,22 +496,399 @@ RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
return self.run_cmd('docker', 'push', image_tag2)
class AmdImageBuilder(Builder):
"""Build ModelScope image on top of vllm/vllm-openai-rocm."""
@staticmethod
def _is_specific_release_tag(tag: str) -> bool:
tag = tag.strip()
if not tag or tag.lower() in _FLOATING_ROCM_TAGS:
return False
if tag.endswith('-base'):
return False
if _NIGHTLY_HASH_PATTERN.fullmatch(tag):
return False
return bool(_VERSION_TAG_PATTERN.fullmatch(tag))
@staticmethod
def _image_digest(tag_info: dict) -> Optional[str]:
digest = tag_info.get('digest')
if digest:
return digest
for image in tag_info.get('images') or []:
digest = image.get('digest')
if digest:
return digest
return None
@classmethod
def _fetch_rocm_tags(cls, page_size: int = 100) -> List[dict]:
tags: List[dict] = []
url = (f'https://hub.docker.com/v2/repositories/{VLLM_ROCM_REPO}/tags'
f'?page_size={page_size}&ordering=-last_updated')
while url:
req = urllib.request.Request(
url, headers={'User-Agent': 'modelscope-docker-builder'})
try:
with urllib.request.urlopen(req, timeout=60) as resp:
payload = json.load(resp)
except (urllib.error.URLError, json.JSONDecodeError) as exc:
raise RuntimeError(
f'Failed to query Docker Hub tags for {VLLM_ROCM_REPO}: '
f'{exc}') from exc
tags.extend(payload.get('results') or [])
url = payload.get('next')
# Only scan the first few pages; release tags are near the top.
if len(tags) >= 300:
break
if not tags:
raise RuntimeError(
f'No tags returned from Docker Hub for {VLLM_ROCM_REPO}')
return tags
@classmethod
def resolve_latest_rocm_tag(cls) -> str:
"""Resolve the newest concrete release tag for vllm-openai-rocm.
Preference order:
1. Semver tag (vX.Y.Z) that shares digest with floating ``latest``
2. Newest semver tag by Docker Hub ``last_updated``
"""
tags = cls._fetch_rocm_tags()
by_name = {item['name']: item for item in tags if item.get('name')}
release_tags = [
item for item in tags
if cls._is_specific_release_tag(item.get('name', ''))
]
latest_info = by_name.get('latest')
latest_digest = cls._image_digest(latest_info) if latest_info else None
if latest_digest:
matched = [
item for item in release_tags
if cls._image_digest(item) == latest_digest
]
if matched:
# Prefer the first match in last_updated order from API.
chosen = matched[0]['name']
print(
f'Resolved {VLLM_ROCM_REPO} latest digest to release tag: '
f'{chosen}')
return chosen
if not release_tags:
raise RuntimeError(
f'No concrete release tags found for {VLLM_ROCM_REPO}')
chosen = release_tags[0]['name']
print(f'Resolved newest {VLLM_ROCM_REPO} release tag: {chosen}')
return chosen
def init_args(self, args: Any) -> Any:
# Auto-discover from Docker Hub unless an explicit override is given.
override = getattr(args, 'base_image_tag', None)
if override and str(override).strip() and str(
override).strip().lower() not in {'auto', 'latest'}:
args.base_image_tag = str(override).strip()
if not self._is_specific_release_tag(args.base_image_tag):
raise ValueError(
'base_image_tag override must be a concrete release tag '
f'(e.g. v0.25.1), got: {args.base_image_tag}')
print(f'Using override AMD ROCm base image tag: '
f'{args.base_image_tag}')
else:
args.base_image_tag = self.resolve_latest_rocm_tag()
if not args.base_image:
args.base_image = f'{VLLM_ROCM_REPO}:{args.base_image_tag}'
if not args.cuda_version:
args.cuda_version = '0.0.0'
return args
@staticmethod
def _sanitize_tag(tag: str) -> str:
return re.sub(r'[^A-Za-z0-9._-]+', '-', tag)
@staticmethod
def _normalize_version(version: str) -> str:
version = version.strip().lstrip('vV')
version = version.split('+')[0].split(' ')[0]
return re.sub(r'[^0-9A-Za-z._-]+', '', version)
@staticmethod
def _python_tag_from_version(version: str) -> str:
parts = version.strip().split('.')
if len(parts) >= 2 and parts[0].isdigit() and parts[1].isdigit():
return f'py{parts[0]}{parts[1]}'
return f'py{re.sub(r"[^0-9]", "", version)}'
@classmethod
def _run_capture(cls, *cmd: str) -> subprocess.CompletedProcess:
return subprocess.run(
list(cmd), capture_output=True, text=True, check=False)
@classmethod
def _probe_via_entrypoint(cls, base_image: str) -> dict:
"""Read versions with docker run --entrypoint (no GPU required)."""
# Keep this script compact: it runs inside the base image via python -c.
script = (
'import json,os,pathlib,subprocess,sys\n'
'info={"python":"%d.%d.%d"%sys.version_info[:3]}\n'
'try:\n'
' import torch\n'
' info["torch"]=torch.__version__\n'
' hip=getattr(torch.version,"hip",None)\n'
' if hip: info["torch_hip"]=hip\n'
'except Exception as e:\n'
' info["torch_error"]=str(e)\n'
'for p in ("/opt/rocm/.info/version","/opt/rocm/.info/version-dev"):\n'
' f=pathlib.Path(p)\n'
' if f.is_file():\n'
' info["rocm_file"]=f.read_text().strip().splitlines()[0]\n'
' break\n'
'for k in ("ROCM_VERSION","HIP_VERSION","TORCH_VERSION"):\n'
' if os.environ.get(k): info[k.lower()]=os.environ[k]\n'
'def _dpkg_ver(*names):\n'
' for n in names:\n'
' try:\n'
' r=subprocess.run(["dpkg-query","-W","-f=${Version}",n],'
'capture_output=True,text=True)\n'
' if r.returncode==0 and r.stdout.strip():\n'
' return r.stdout.strip()\n'
' except Exception:\n'
' pass\n'
' return None\n'
'def _dpkg_scan(prefixes):\n'
' try:\n'
' r=subprocess.run(["dpkg-query","-W","-f=${Package}\\t${Version}\\n"],'
'capture_output=True,text=True)\n'
' except Exception:\n'
' return {}\n'
' found={}\n'
' for line in (r.stdout or "").splitlines():\n'
' if "\\t" not in line: continue\n'
' pkg,ver=line.split("\\t",1)\n'
' for pref in prefixes:\n'
' if pkg==pref or pkg.startswith(pref+"-"):\n'
' found.setdefault(pref,ver)\n'
' return found\n'
'pkgs=_dpkg_scan(("rccl","miopen"))\n'
'info["system.library.rccl"]=_dpkg_ver("rccl") or pkgs.get("rccl")\n'
'info["system.library.miopen"]=('
'_dpkg_ver("miopen-hip","miopen") or pkgs.get("miopen"))\n'
'print(json.dumps(info))\n')
for py in ('python3', 'python'):
result = cls._run_capture('docker', 'run', '--rm', '--network',
'none', '--entrypoint', py, base_image,
'-c', script)
if result.returncode == 0 and result.stdout.strip():
try:
return json.loads(result.stdout.strip().splitlines()[-1])
except json.JSONDecodeError:
continue
return {}
@classmethod
def _probe_via_history(cls, base_image: str) -> dict:
"""Parse build ARGs from docker history (no container start)."""
result = cls._run_capture('docker', 'history', '--no-trunc',
'--format', '{{.CreatedBy}}', base_image)
if result.returncode != 0:
return {}
text = result.stdout
info = {}
for key, pattern in (
('rocm', r'ROCM_VERSION=([0-9]+(?:\.[0-9]+)*)'),
('python', r'PYTHON_VERSION=([0-9]+(?:\.[0-9]+)*)'),
('ubuntu',
r'org\.opencontainers\.image\.version=([0-9]+(?:\.[0-9]+)*)'),
):
matches = re.findall(pattern, text)
if matches:
# docker history lists newest layers first.
info[key] = matches[0]
return info
@classmethod
def _probe_via_create_cp(cls, base_image: str) -> dict:
"""Copy version files out of a created (not started) container."""
import tempfile
create = cls._run_capture('docker', 'create', base_image)
if create.returncode != 0:
return {}
cid = create.stdout.strip()
info = {}
try:
with tempfile.TemporaryDirectory() as tmp:
dest = os.path.join(tmp, 'version')
for src in ('/opt/rocm/.info/version',
'/opt/rocm/.info/version-dev'):
result = cls._run_capture('docker', 'cp', f'{cid}:{src}',
dest)
if result.returncode == 0 and os.path.isfile(dest):
with open(dest, 'r', encoding='utf-8') as f:
line = f.read().strip().splitlines()
if line:
info['rocm_file'] = line[0].strip()
break
finally:
cls._run_capture('docker', 'rm', '-f', cid)
return info
@classmethod
def probe_base_image_versions(cls, base_image: str) -> dict:
"""Discover rocm/python/torch without needing AMD GPU.
Methods (in order):
1. docker run --entrypoint python -c ... (CPU-only, no --device)
2. docker history --no-trunc parse ROCM_VERSION/PYTHON_VERSION
3. docker create + docker cp /opt/rocm/.info/version
"""
probed = {}
entry = cls._probe_via_entrypoint(base_image)
history = cls._probe_via_history(base_image)
copied = cls._probe_via_create_cp(base_image)
probed.update(history)
probed.update(copied)
probed.update(entry)
rocm = (
probed.get('rocm_file') or probed.get('rocm_version')
or probed.get('rocm') or probed.get('torch_hip')
or probed.get('hip_version'))
python_ver = probed.get('python')
torch_ver = probed.get('torch') or probed.get('torch_version')
ubuntu_ver = probed.get('ubuntu')
# Keep dpkg package versions as-is (may contain '~', e.g. 2.27.7.70201-81~22.04).
rccl_ver = probed.get('system.library.rccl')
miopen_ver = probed.get('system.library.miopen')
versions = {
'rocm': cls._normalize_version(rocm) if rocm else None,
'python':
cls._normalize_version(python_ver) if python_ver else None,
'torch': cls._normalize_version(torch_ver) if torch_ver else None,
'ubuntu':
cls._normalize_version(ubuntu_ver) if ubuntu_ver else None,
'system.library.rccl': rccl_ver or None,
'system.library.miopen': miopen_ver or None,
}
print('Probed AMD base image versions:')
for key, value in versions.items():
print(f' {key}: {value or "unknown"}')
return versions
def generate_dockerfile(self) -> str:
with open('docker/Dockerfile.amd', 'r') as f:
content = f.read()
content = content.replace('{base_image}', self.args.base_image)
content = content.replace('{base_image_tag}', self.args.base_image_tag)
content = content.replace('{modelscope_branch}',
self.args.modelscope_branch)
content = content.replace('{cur_time}', formatted_time)
return content
def image(self) -> str:
ubuntu = getattr(self.args, 'amd_ubuntu_version',
None) or self.args.ubuntu_version
rocm = getattr(self.args, 'amd_rocm_version', None)
py_tag = getattr(self.args, 'amd_python_tag', None) or getattr(
self.args, 'python_tag', None)
torch = getattr(self.args, 'amd_torch_version', None)
if not (rocm and py_tag and torch):
raise RuntimeError(
'AMD image tag requires probed rocm/python/torch versions. '
f'Got rocm={rocm}, python={py_tag}, torch={torch}')
return (f'{docker_registry}:ubuntu{ubuntu}-rocm{rocm}-{py_tag}-'
f'torch{torch}-{self.args.modelscope_version}-test')
def _log_base_image_info(self) -> int:
base_image = self.args.base_image
print('=' * 60)
print(f'AMD ROCm base image: {base_image}')
print(f'AMD ROCm base image tag: {self.args.base_image_tag}')
print('=' * 60)
ret = self.run_cmd('docker', 'pull', base_image)
if ret != 0:
return ret
result = self._run_capture(
'docker', 'image', 'inspect', base_image,
'--format={{.Id}} {{if index .RepoDigests 0}}'
'{{index .RepoDigests 0}}{{else}}local-only{{end}}')
if result.returncode == 0:
print(f'AMD base image resolved: {result.stdout.strip()}')
else:
print(f'AMD base image inspect warning: {result.stderr.strip()}')
versions = self.probe_base_image_versions(base_image)
if not versions.get('rocm') or not versions.get(
'python') or not versions.get('torch'):
print('ERROR: failed to probe rocm/python/torch from base image')
return 1
self.args.amd_rocm_version = versions['rocm']
self.args.amd_torch_version = versions['torch']
self.args.amd_python_tag = self._python_tag_from_version(
versions['python'])
if versions.get('ubuntu'):
self.args.amd_ubuntu_version = versions['ubuntu']
else:
self.args.amd_ubuntu_version = self.args.ubuntu_version
print(f'AMD output image tag will be: {self.image()}')
print('=' * 60)
return 0
def build(self) -> int:
ret = self._log_base_image_info()
if ret != 0:
return ret
return self.run_cmd('docker', 'build', '-t', self.image(), '-f',
'Dockerfile', '.')
def push(self):
image_name = self.image()
ret = self.run_cmd('docker', 'push', image_name)
if ret != 0:
return ret
ubuntu = self.args.amd_ubuntu_version
rocm = self.args.amd_rocm_version
py_tag = self.args.amd_python_tag
torch = self.args.amd_torch_version
image_tag2 = (f'{docker_registry}:ubuntu{ubuntu}-rocm{rocm}-{py_tag}-'
f'torch{torch}-{self.args.modelscope_version}-'
f'{formatted_time}-test')
ret = self.run_cmd('docker', 'tag', image_name, image_tag2)
if ret != 0:
return ret
print(f'AMD image timestamp tag: {image_tag2}')
return self.run_cmd('docker', 'push', image_tag2)
class AscendImageBuilder(StableGPUImageBuilder):
_DEFAULT_TORCH_VERSION = '2.9.0'
_DEFAULT_TORCHVISION_VERSION = '0.24.0'
_DEFAULT_TORCHAUDIO_VERSION = '2.9.0'
_DEFAULT_TORCH_NPU_VERSION = '2.9.0.post2'
_DEFAULT_VLLM_VERSION = '0.18.0'
_DEFAULT_VLLM_ASCEND_VERSION = '0.18.0'
_DEFAULT_TRITON_ASCEND_VERSIONS = {
'8.5': '3.2.0',
'9.0': '3.2.1',
}
_CANN_VERSION_PATTERN = re.compile(r'^\d+(?:\.[0-9A-Za-z]+)+$')
_OS_TAG_PATTERN = re.compile(r'^[A-Za-z]+[0-9][0-9A-Za-z.]*$')
_PYTHON_TAG_PATTERN = re.compile(r'^py\d+\.\d+$', re.IGNORECASE)
_TORCH_NPU_VERSION_PATTERN = re.compile(
r'^(?P<torch_version>\d+\.\d+\.\d+)(?:\.post\d+)?$')
@staticmethod
def _normalize_arch(arch: str = None) -> str:
arch = arch or platform.machine()
arch = arch.lower()
arch_mapping = {
'x86': 'x86',
'x86_64': 'x86',
'amd64': 'x86',
'arm': 'arm',
'aarch64': 'arm',
'arm64': 'arm',
'x86': 'x86_64',
'x86_64': 'x86_64',
'amd64': 'x86_64',
'arm': 'aarch64',
'aarch64': 'aarch64',
'arm64': 'aarch64',
}
if arch not in arch_mapping:
raise ValueError(f'Unsupported architecture: {arch}. '
@@ -536,34 +928,130 @@ class AscendImageBuilder(StableGPUImageBuilder):
cann_version = parts[0]
os_tag = parts[2]
python_tag = parts[3]
if not cls._CANN_VERSION_PATTERN.fullmatch(cann_version):
raise ValueError(f'Invalid CANN version in Ascend base image tag: '
f'{cann_version}')
if not cls._OS_TAG_PATTERN.fullmatch(os_tag):
raise ValueError(
f'Invalid OS tag in Ascend base image tag: {os_tag}')
if not cls._PYTHON_TAG_PATTERN.fullmatch(python_tag):
raise ValueError(
f'Invalid Python tag in Ascend base image tag: {python_tag}')
return cann_version, f'CANN{cann_version}', os_tag
return cann_version, f'CANN{cann_version}', os_tag, python_tag
@staticmethod
def _get_os_family(os_tag: str) -> str:
os_tag = os_tag.lower()
if os_tag.startswith('ubuntu'):
return 'ubuntu'
if os_tag.startswith('openeuler'):
return 'openeuler'
raise ValueError(f'Unsupported Ascend base image OS tag: {os_tag}. '
'Supported OS families are Ubuntu and openEuler.')
@classmethod
def _init_torch_versions(cls, args) -> None:
torch_version_specified = args.torch_version is not None
torchvision_version_specified = args.torchvision_version is not None
torchaudio_version_specified = args.torchaudio_version is not None
if torch_version_specified:
if (not torchvision_version_specified
or not torchaudio_version_specified):
raise ValueError(
'When overriding --torch_version for an Ascend image, also '
'pass matching --torchvision_version and '
'--torchaudio_version.')
elif torchvision_version_specified or torchaudio_version_specified:
raise ValueError(
'--torchvision_version and --torchaudio_version require an '
'explicit --torch_version for an Ascend image.')
args.torch_version = args.torch_version or cls._DEFAULT_TORCH_VERSION
args.torchvision_version = (
args.torchvision_version or cls._DEFAULT_TORCHVISION_VERSION)
args.torchaudio_version = (
args.torchaudio_version or cls._DEFAULT_TORCHAUDIO_VERSION)
args.torch_npu_version = (
args.torch_npu_version or cls._DEFAULT_TORCH_NPU_VERSION)
match = cls._TORCH_NPU_VERSION_PATTERN.fullmatch(
args.torch_npu_version)
if not match:
raise ValueError('Invalid --torch_npu_version. Expected '
'<major>.<minor>.<patch> or '
'<major>.<minor>.<patch>.post<revision>.')
if args.torch_version != match.group('torch_version'):
raise ValueError(
'--torch_version must exactly match the base version of '
f'--torch_npu_version, got torch={args.torch_version} and '
f'torch_npu={args.torch_npu_version}.')
@classmethod
def _init_component_versions(cls, args) -> None:
args.vllm_version = args.vllm_version or cls._DEFAULT_VLLM_VERSION
args.vllm_ascend_version = (
args.vllm_ascend_version or cls._DEFAULT_VLLM_ASCEND_VERSION)
args.vllm_git_ref = cls._get_vllm_git_ref(args.vllm_version)
args.vllm_ascend_git_ref = cls._get_vllm_git_ref(
args.vllm_ascend_version)
if not args.triton_ascend_version:
cann_series = '.'.join(args.cann_version.split('.')[:2])
try:
args.triton_ascend_version = (
cls._DEFAULT_TRITON_ASCEND_VERSIONS[cann_series])
except KeyError as e:
raise ValueError('No default triton-ascend version for CANN '
f'{args.cann_version}. Please pass '
'--triton_ascend_version explicitly.') from e
@staticmethod
def _get_vllm_git_ref(version: str) -> str:
return version if version.startswith('v') else f'v{version}'
def init_args(self, args) -> Any:
if not args.base_image:
# Reuse the prebuilt vllm-ascend image to avoid rebuilding its stack.
args.base_image = 'quay.io/ascend/cann:8.5.1-a3-ubuntu22.04-py3.11'
self._init_torch_versions(args)
args.arch = self._normalize_arch(args.arch)
args.atlas_hardware = self._get_atlas_hardware(args.soc_version)
args.cann_version, args.cann_version_tag, args.os_tag = (
self._get_cann_os_tags(args.base_image))
(args.cann_version, args.cann_version_tag, args.os_tag,
args.ascend_python_tag) = (
self._get_cann_os_tags(args.base_image))
self._get_os_family(args.os_tag)
self._init_component_versions(args)
return super().init_args(args)
def _generate_python_tag(self, _python_version: str) -> str:
return self.args.ascend_python_tag
def generate_dockerfile(self) -> str:
extra_content = """
RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
RUN export PIP_EXTRA_INDEX_URL=https://pypi.org/simple && \
pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
"""
with open('docker/Dockerfile.ascend', 'r') as f:
content = f.read()
content = content.replace('{base_image}', self.args.base_image)
content = content.replace('{soc_version}', self.args.soc_version)
content = content.replace('{cann_version}', self.args.cann_version)
content = content.replace('{torch_version}',
self.args.torch_version)
content = content.replace('{torchvision_version}',
self.args.torchvision_version)
content = content.replace('{torchaudio_version}',
self.args.torchaudio_version)
content = content.replace('{torch_npu_version}',
self.args.torch_npu_version)
content = content.replace('{vllm_git_ref}', self.args.vllm_git_ref)
content = content.replace('{vllm_ascend_git_ref}',
self.args.vllm_ascend_git_ref)
content = content.replace('{triton_ascend_version}',
self.args.triton_ascend_version)
content = content.replace('{extra_content}', extra_content)
content = content.replace('{cur_time}', formatted_time)
content = content.replace('{install_ms_deps}', 'False')
@@ -577,11 +1065,12 @@ RUN pip install --no-cache-dir -U icecream soundfile pybind11 py-spy
return content
def image(self) -> str:
return (
f'{docker_registry}:{self.args.swift_branch}-'
f'{self.args.atlas_hardware}-{self.args.python_tag}-'
f'{self.args.cann_version_tag}-{self.args.os_tag}-{self.args.arch}'
)
tag = (f'{self.args.swift_branch}-{self.args.cann_version_tag}-'
f'torch_npu{self.args.torch_npu_version}-'
f'{self.args.atlas_hardware}-{self.args.os_tag}-'
f'{self.args.python_tag}-'
f'{self.args.arch}')
return f'{docker_registry}:{tag.lower()}'
def push(self):
return 0
@@ -593,6 +1082,7 @@ parser.add_argument('--image_type', type=str)
parser.add_argument('--python_version', type=str, default='3.12.13')
parser.add_argument('--ubuntu_version', type=str, default='22.04')
parser.add_argument('--torch_version', type=str, default=None)
parser.add_argument('--torch_npu_version', type=str, default=None)
parser.add_argument('--torchvision_version', type=str, default=None)
parser.add_argument('--cuda_version', type=str, default=None)
parser.add_argument('--ci_image', type=int, default=0)
@@ -600,6 +1090,8 @@ parser.add_argument('--torchaudio_version', type=str, default=None)
parser.add_argument('--optimum_version', type=str, default=None)
parser.add_argument('--tf_version', type=str, default=None)
parser.add_argument('--vllm_version', type=str, default=None)
parser.add_argument('--vllm_ascend_version', type=str, default=None)
parser.add_argument('--triton_ascend_version', type=str, default=None)
parser.add_argument('--lmdeploy_version', type=str, default=None)
parser.add_argument('--flashattn_version', type=str, default=None)
parser.add_argument('--autogptq_version', type=str, default=None)
@@ -610,6 +1102,12 @@ parser.add_argument('--megatron_branch', type=str, default='v0.15.3')
parser.add_argument('--mindspeed_branch', type=str, default='core_r0.15.3')
parser.add_argument('--soc_version', type=str, default='ascend910_9391')
parser.add_argument('--arch', type=str, choices=['x86', 'arm'], default=None)
parser.add_argument(
'--base_image_tag',
type=str,
default=None,
help='Optional AMD ROCm override tag. Default: auto-resolve newest '
'concrete vllm/vllm-openai-rocm release tag from Docker Hub.')
parser.add_argument('--dry_run', type=int, default=0)
args = parser.parse_args()
@@ -621,6 +1119,8 @@ elif args.image_type.lower() == 'stable':
builder_cls = [StableCPUImageBuilder, StableGPUImageBuilder]
elif args.image_type.lower() == 'ascend':
builder_cls = [AscendImageBuilder]
elif args.image_type.lower() == 'amd':
builder_cls = [AmdImageBuilder]
elif args.image_type.lower() == 'latest':
builder_cls = [LatestGPUImageBuilder]
else:

View File

@@ -47,5 +47,4 @@ else
fi
pip config set global.index-url https://mirrors.cloud.aliyuncs.com/pypi/simple
pip config set global.extra-index-url https://pypi.org/simple
pip config set install.trusted-host mirrors.cloud.aliyuncs.com