diff --git a/onnxruntime/core/providers/rocm/rocm_execution_provider.cc b/onnxruntime/core/providers/rocm/rocm_execution_provider.cc index be6f0efc27..f05e0f949b 100644 --- a/onnxruntime/core/providers/rocm/rocm_execution_provider.cc +++ b/onnxruntime/core/providers/rocm/rocm_execution_provider.cc @@ -57,16 +57,30 @@ ONNX_OPERATOR_KERNEL_EX( } // namespace rocm -ROCMExecutionProvider::PerThreadContext::PerThreadContext(OrtDevice::DeviceId device_id) { +ROCMExecutionProvider::PerThreadContext::PerThreadContext(OrtDevice::DeviceId device_id, size_t hip_mem_limit, ArenaExtendStrategy arena_extend_strategy) { HIP_CALL_THROW(hipSetDevice(device_id)); ROCBLAS_CALL_THROW(rocblas_create_handle(&rocblas_handle_)); MIOPEN_CALL_THROW(miopenCreate(&miopen_handle_)); + + AllocatorCreationInfo default_memory_info( + [](OrtDevice::DeviceId id) { + return onnxruntime::make_unique(id, CUDA); + }, + device_id, + true, + {hip_mem_limit, + static_cast(arena_extend_strategy), + -1, -1}); + + // HIP malloc/free is expensive so always use an arena + allocator_ = CreateAllocator(default_memory_info); } ROCMExecutionProvider::PerThreadContext::~PerThreadContext() { // dtor shouldn't throw. if something went wrong earlier (e.g. out of HIP memory) the handles // here may be bad, and the destroy calls can throw. // https://isocpp.github.io/CppCoreGuidelines/CppCoreGuidelines#Rc-dtor-noexcept + try { ROCBLAS_CALL(rocblas_destroy_handle(rocblas_handle_)); } catch (const std::exception& ex) { @@ -200,7 +214,7 @@ ROCMExecutionProvider::PerThreadContext& ROCMExecutionProvider::GetPerThreadCont // get or create a context if (context_state_.retired_context_pool.empty()) { - context = std::make_shared(device_id_); + context = std::make_shared(device_id_, hip_mem_limit_, arena_extend_strategy_); } else { context = context_state_.retired_context_pool.back(); context_state_.retired_context_pool.pop_back(); @@ -236,6 +250,17 @@ void ROCMExecutionProvider::ReleasePerThreadContext() const { per_thread_context_cache->erase(cached_context_it); } +AllocatorPtr ROCMExecutionProvider::GetAllocator(int id, OrtMemType mem_type) const { + // Pinned memory allocator is shared between threads, but HIP memory allocator is per-thread or it may cause result changes + // A hypothesis is that arena allocator is not aligned with HIP output cache, and data from different kernel writes may + // cause cacheline to contain dirty data. + if (mem_type == OrtMemTypeDefault) { + return GetPerThreadContext().GetAllocator(); + } else { + return IExecutionProvider::GetAllocator(id, mem_type); + } +} + Status ROCMExecutionProvider::Sync() const { HIP_RETURN_IF_ERROR(hipDeviceSynchronize()); return Status::OK(); diff --git a/onnxruntime/core/providers/rocm/rocm_execution_provider.h b/onnxruntime/core/providers/rocm/rocm_execution_provider.h index 2f628bfb6e..e026c0dccd 100644 --- a/onnxruntime/core/providers/rocm/rocm_execution_provider.h +++ b/onnxruntime/core/providers/rocm/rocm_execution_provider.h @@ -32,6 +32,8 @@ class ROCMExecutionProvider : public IExecutionProvider { explicit ROCMExecutionProvider(const ROCMExecutionProviderInfo& info); virtual ~ROCMExecutionProvider(); + AllocatorPtr GetAllocator(int id, OrtMemType mem_type) const override; + Status Sync() const override; Status OnRunStart() override; @@ -53,24 +55,7 @@ class ROCMExecutionProvider : public IExecutionProvider { template const T* GetConstOnes(size_t count) { - if (std::is_same::value) { - if (!constant_ones_float_) { - constant_ones_float_ = rocm::CreateConstantOnes(); - } - return reinterpret_cast(constant_ones_float_->GetBuffer(count)); - } else if (std::is_same::value) { - if (!constant_ones_double_) { - constant_ones_double_ = rocm::CreateConstantOnes(); - } - return reinterpret_cast(constant_ones_double_->GetBuffer(count)); - } else if (std::is_same::value) { - if (!constant_ones_half_) { - constant_ones_half_ = rocm::CreateConstantOnes(); - } - return reinterpret_cast(constant_ones_half_->GetBuffer(count)); - } else { - return nullptr; - } + return GetPerThreadContext().template GetConstOnes(count); } void AddDeferredReleaseCPUPtr(void* p); @@ -108,13 +93,9 @@ class ROCMExecutionProvider : public IExecutionProvider { std::unordered_map deferred_release_cpu_ptr_; OrtMutex deferred_release_cpu_ptr_mutex_; - std::unique_ptr> constant_ones_float_; - std::unique_ptr> constant_ones_double_; - std::unique_ptr> constant_ones_half_; - class PerThreadContext final { public: - PerThreadContext(OrtDevice::DeviceId device_id); + PerThreadContext(OrtDevice::DeviceId device_id, size_t hip_mem_limit, ArenaExtendStrategy arena_extend_strategy); ~PerThreadContext(); rocblas_handle RocblasHandle() const { @@ -129,6 +110,32 @@ class ROCMExecutionProvider : public IExecutionProvider { return current_deferred_release_event_; } + template + const T* GetConstOnes(size_t count) { + if (std::is_same::value) { + if (!constant_ones_float_) { + constant_ones_float_ = rocm::CreateConstantOnes(); + } + return reinterpret_cast(constant_ones_float_->GetBuffer(count)); + } else if (std::is_same::value) { + if (!constant_ones_double_) { + constant_ones_double_ = rocm::CreateConstantOnes(); + } + return reinterpret_cast(constant_ones_double_->GetBuffer(count)); + } else if (std::is_same::value) { + if (!constant_ones_half_) { + constant_ones_half_ = rocm::CreateConstantOnes(); + } + return reinterpret_cast(constant_ones_half_->GetBuffer(count)); + } else { + return nullptr; + } + } + + AllocatorPtr GetAllocator() const { + return allocator_; + } + private: rocblas_handle rocblas_handle_ = nullptr; miopenHandle_t miopen_handle_ = nullptr; @@ -137,6 +144,12 @@ class ROCMExecutionProvider : public IExecutionProvider { // note that hipEvent will be assigned at OnRunEnd() when PerThreadContext destory // so the ownership is passed to deferred_release_cpu_ptr_ hipEvent_t current_deferred_release_event_ = nullptr; + + std::unique_ptr> constant_ones_float_; + std::unique_ptr> constant_ones_double_; + std::unique_ptr> constant_ones_half_; + + AllocatorPtr allocator_; }; using PerThreadContextMap = std::unordered_map>; diff --git a/orttraining/tools/amdgpu/Dockerfile.rocm3.7 b/orttraining/tools/amdgpu/Dockerfile.rocm3.7 index a85254a56d..0013b522b6 100644 --- a/orttraining/tools/amdgpu/Dockerfile.rocm3.7 +++ b/orttraining/tools/amdgpu/Dockerfile.rocm3.7 @@ -120,7 +120,7 @@ RUN CC=mpicc MPICC=mpicc pip install mpi4py --no-binary mpi4py # ONNX Runtime WORKDIR $GITHUB_DIR ENV ORT_DIR=$GITHUB_DIR/onnxruntime -RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxruntime.git \ +RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \ && cd onnxruntime \ && python3 tools/ci_build/build.py \ --cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \ diff --git a/orttraining/tools/amdgpu/Dockerfile.rocm3.7.pytorch b/orttraining/tools/amdgpu/Dockerfile.rocm3.7.pytorch index fb5aad6abb..1a019f805f 100644 --- a/orttraining/tools/amdgpu/Dockerfile.rocm3.7.pytorch +++ b/orttraining/tools/amdgpu/Dockerfile.rocm3.7.pytorch @@ -114,7 +114,7 @@ ARG CACHE_DATA=2020-10-28 # ONNX Runtime WORKDIR $GITHUB_DIR ENV ORT_DIR=$GITHUB_DIR/onnxruntime -RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxruntime.git \ +RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \ && cd onnxruntime \ && python3 tools/ci_build/build.py \ --cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \ @@ -126,7 +126,7 @@ RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxrunt --use_rocm --rocm_home /opt/rocm \ --mpi_home $OPENMPI_DIR \ --nccl_home /opt/rocm \ - --enable_training \ + --enable_training \ && test -f $ORT_DIR/build/RelWithDebInfo/onnxruntime_training_bert \ && pip install $ORT_DIR/build/RelWithDebInfo/dist/*.whl \ && ldconfig @@ -157,7 +157,7 @@ RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training && cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \ && python3 -m pip install --no-cache-dir -e . \ && python3 -m pip install --no-cache-dir -r examples/requirements.txt \ - && python3 -m pip install cerberus \ + && python3 -m pip install cerberus sympy \ && cd .. \ && wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \ && unzip ${GPT2_DATASET}-v1.zip diff --git a/orttraining/tools/amdgpu/Dockerfile.rocm3.8.pytorch b/orttraining/tools/amdgpu/Dockerfile.rocm3.8.pytorch index a40dd3e36b..f1fe42da37 100644 --- a/orttraining/tools/amdgpu/Dockerfile.rocm3.8.pytorch +++ b/orttraining/tools/amdgpu/Dockerfile.rocm3.8.pytorch @@ -113,7 +113,7 @@ ARG CACHE_DATA=2020-10-28 # ONNX Runtime WORKDIR $GITHUB_DIR ENV ORT_DIR=$GITHUB_DIR/onnxruntime -RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxruntime.git \ +RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \ && cd onnxruntime \ && python3 tools/ci_build/build.py \ --cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \ @@ -156,7 +156,7 @@ RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training && cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \ && python3 -m pip install --no-cache-dir -e . \ && python3 -m pip install --no-cache-dir -r examples/requirements.txt \ - && python3 -m pip install cerberus \ + && python3 -m pip install cerberus sympy \ && cd .. \ && wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \ && unzip ${GPT2_DATASET}-v1.zip diff --git a/orttraining/tools/amdgpu/Dockerfile.rocm3.9 b/orttraining/tools/amdgpu/Dockerfile.rocm3.9 new file mode 100644 index 0000000000..7fdd00b712 --- /dev/null +++ b/orttraining/tools/amdgpu/Dockerfile.rocm3.9 @@ -0,0 +1,196 @@ +# docker build --network=host --file Dockerfile.rocm3.9 --tag ort:rocm3.9-ort-dev . + +FROM rocm/tensorflow:rocm3.9-tf2.3-dev + +RUN wget -q -O - http://repo.radeon.com/rocm/rocm.gpg.key | apt-key add - +RUN echo 'deb [arch=amd64] http://repo.radeon.com/rocm/apt/3.9/ xenial main' | tee /etc/apt/sources.list.d/rocm.list + +RUN apt-get -y update +RUN apt-get -y install apt-utils +RUN apt-get -y install build-essential autotools-dev \ + make git curl vim wget rsync jq openssh-server openssh-client sudo \ + iputils-ping net-tools ethtool libcap2 \ + automake autoconf libtool flex doxygen \ + perl lsb-release iproute2 pciutils graphviz \ + bc tar git bash pbzip2 pv bzip2 cabextract \ + g++ gcc \ + && apt-get autoremove + +# sh +RUN rm /bin/sh && ln -s /bin/bash /bin/sh + +# Labels for the docker +LABEL description="This docker sets up the environment to run ORT Training with AMD GPU" + +# CMake +ENV CMAKE_VERSION=3.18.2 +RUN cd /usr/local && \ + wget -q -O - https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-Linux-x86_64.tar.gz | tar zxf - +ENV PATH=/usr/local/cmake-${CMAKE_VERSION}-Linux-x86_64/bin:${PATH} + +# WORKSPACE_DIR +ENV WORKSPACE_DIR=/workspace +RUN mkdir -p $WORKSPACE_DIR +WORKDIR $WORKSPACE_DIR + +# Infiniband setup, openmpi installed under /usr/mpi/gcc/openmpi-4.0.4rc3 doesn't support multi-thread +ENV MOFED_VERSION=5.1-0.6.6.0 +ENV MOFED_OS=ubuntu18.04 +ENV MOFED_FILENAME=MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 +RUN curl -fSsL https://www.mellanox.com/downloads/ofed/MLNX_OFED-${MOFED_VERSION}/${MOFED_FILENAME}.tgz | tar -zxpf - +RUN cd MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 && \ + ./mlnxofedinstall --force --user-space-only --without-fw-update --hpc && \ + cd .. && \ + rm -r MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 + +# install miniconda (comes with python 3.9 default) +ARG CONDA_VERSION=4.7.10 +ARG CONDA_URL=https://repo.anaconda.com/miniconda/Miniconda3-${CONDA_VERSION}-Linux-x86_64.sh +RUN curl -fSsL --insecure ${CONDA_URL} -o install-conda.sh &&\ + /bin/bash ./install-conda.sh -b -p /opt/conda &&\ + /opt/conda/bin/conda clean -ya +ENV PATH=/opt/conda/bin:${PATH} + +ARG NUMPY_VERSION=1.18.5 +ARG ONNX_VERSION=1.7.0 +RUN conda install -y \ + numpy=${NUMPY_VERSION} \ + cmake \ + ninja \ + pyyaml \ + cffi \ + setuptools \ + && pip install --no-cache-dir wheel tqdm boto3 requests six ipdb h5py html2text nltk progressbar \ + git+https://github.com/NVIDIA/dllogger \ + onnx=="${ONNX_VERSION}" + +# GITHUB_DIR +ENV GITHUB_DIR=$WORKSPACE_DIR/github +RUN mkdir -p $GITHUB_DIR + +# UCX +WORKDIR $GITHUB_DIR +RUN apt-get -y update && apt-get -y --no-install-recommends install libnuma-dev +ARG UCX_VERSION=1.9.0-rc3 +ENV UCX_DIR=$WORKSPACE_DIR/ucx-$UCX_VERSION +RUN git clone https://github.com/openucx/ucx.git \ + && cd ucx \ + && git checkout v$UCX_VERSION \ + && ./autogen.sh \ + && mkdir build \ + && cd build \ + && ../contrib/configure-opt --prefix=$UCX_DIR --without-rocm --without-knem --without-cuda \ + && make -j"$(nproc)" \ + && make install + +# OpenMPI +# note: require --enable-orterun-prefix-by-default for Azure machine learning compute +# note: disable verbs as we use ucx middleware and don't want btl openib warnings +WORKDIR $GITHUB_DIR +ARG OPENMPI_BASEVERSION=4.0 +ARG OPENMPI_VERSION=${OPENMPI_BASEVERSION}.5 +ENV OPENMPI_DIR=$WORKSPACE_DIR/openmpi-${OPENMPI_VERSION} +RUN git clone --recursive https://github.com/open-mpi/ompi.git \ + && cd ompi \ + && git checkout v$OPENMPI_VERSION \ + && ./autogen.pl \ + && mkdir build \ + && cd build \ + && ../configure --prefix=$OPENMPI_DIR --with-ucx=$UCX_DIR --without-verbs \ + --enable-mpirun-prefix-by-default --enable-orterun-prefix-by-default \ + --enable-mca-no-build=btl-uct --disable-mpi-fortran \ + && make -j"$(nproc)" \ + && make install \ + && ldconfig \ + && test -f ${OPENMPI_DIR}/bin/mpic++ + +ENV PATH=$OPENMPI_DIR/bin:${PATH} +ENV LD_LIBRARY_PATH=$OPENMPI_DIR/lib:${LD_LIBRARY_PATH} + +# Create a wrapper for OpenMPI to allow running as root by default +RUN mv $OPENMPI_DIR/bin/mpirun $OPENMPI_DIR/bin/mpirun.real && \ + echo '#!/bin/bash' > $OPENMPI_DIR/bin/mpirun && \ + echo 'mpirun.real --allow-run-as-root "$@"' >> $OPENMPI_DIR/bin/mpirun && \ + chmod a+x $OPENMPI_DIR/bin/mpirun + +# install mpi4py (be sure to link existing /opt/openmpi-xxx) +RUN CC=mpicc MPICC=mpicc pip install mpi4py --no-binary mpi4py + +# ONNX Runtime +WORKDIR $GITHUB_DIR +ENV ORT_DIR=$GITHUB_DIR/onnxruntime +RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \ + && cd onnxruntime \ + && python3 tools/ci_build/build.py \ + --cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \ + --build_dir build \ + --config RelWithDebInfo \ + --parallel \ + --skip_tests \ + --build_wheel \ + --use_rocm --rocm_home /opt/rocm \ + --mpi_home $OPENMPI_DIR \ + --nccl_home /opt/rocm \ + --enable_training \ + && test -f $ORT_DIR/build/RelWithDebInfo/onnxruntime_training_bert \ + && pip install $ORT_DIR/build/RelWithDebInfo/dist/*.whl \ + && ldconfig + +# Instructions to pull and install the nightly ROCm3.8 PyTorch whl pacakge +RUN pip3 install --pre torch -f https://download.pytorch.org/whl/nightly/rocm3.9/torch_nightly.html + +# ONNX Runtime Training Examples +WORKDIR $GITHUB_DIR +ARG GPT2_DATASET=wikitext-103 +RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training-examples.git \ + && cd onnxruntime-training-examples \ + # Nvidia BERT + && git clone --no-checkout https://github.com/NVIDIA/DeepLearningExamples.git \ + && cd DeepLearningExamples \ + && git checkout cf54b787 \ + && cd .. \ + && mv DeepLearningExamples/PyTorch/LanguageModeling/BERT ${WORKSPACE_DIR} \ + && rm -rf DeepLearningExamples \ + && cp -r ./nvidia-bert/ort_addon/* ${WORKSPACE_DIR}/BERT \ + # GPT2 fine-tuning + && cd huggingface-gpt2 \ + && git clone https://github.com/huggingface/transformers.git \ + && cd transformers \ + && git checkout 9a0a8c1c6f4f2f0c80ff07d36713a3ada785eec5 \ + && cd .. \ + && mkdir -p ${WORKSPACE_DIR}/GPT2 \ + && cp -r transformers ${WORKSPACE_DIR}/GPT2 \ + && cd ${WORKSPACE_DIR}/GPT2/transformers \ + && git apply $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/src_changes.patch \ + && cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \ + && python3 -m pip install --no-cache-dir -e . \ + && python3 -m pip install --no-cache-dir -r examples/requirements.txt \ + && python3 -m pip install cerberus sympy packaging \ + && cd .. \ + && wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \ + && unzip ${GPT2_DATASET}-v1.zip + +ENV BERT_DIR=${WORKSPACE_DIR}/BERT +ENV GPT2_DIR=${WORKSPACE_DIR}/GPT2 +ENV TRAIN_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.train.tokens +ENV TEST_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.test.tokens + +# Enable ssh access without password needed +RUN sed -i 's/#PermitRootLogin prohibit-password/PermitRootLogin yes/g' /etc/ssh/sshd_config +RUN sed -i 's/#StrictModes yes/StrictModes no/g' /etc/ssh/sshd_config +RUN sed -i 's/#PubkeyAuthentication yes/PubkeyAuthentication yes/g' /etc/ssh/sshd_config +RUN sed -i 's/#PermitEmptyPasswords no/PermitEmptyPasswords yes/g' /etc/ssh/sshd_config + +# Start or Restart sshd service +ENTRYPOINT service ssh restart && /bin/bash + +# Add model and scripts +ADD model ${WORKSPACE_DIR}/model +ADD script ${WORKSPACE_DIR}/script +RUN chmod a+x ${WORKSPACE_DIR}/script/run_bert.sh + +# add locale en_US.UTF-8 +RUN apt-get install -y locales +RUN locale-gen en_US.UTF-8 + +WORKDIR ${WORKSPACE_DIR}/script diff --git a/orttraining/tools/amdgpu/Dockerfile.rocm3.9.pytorch b/orttraining/tools/amdgpu/Dockerfile.rocm3.9.pytorch new file mode 100644 index 0000000000..9a38f28543 --- /dev/null +++ b/orttraining/tools/amdgpu/Dockerfile.rocm3.9.pytorch @@ -0,0 +1,190 @@ +# docker build --network=host --file Dockerfile.rocm3.9.pytorch --tag ort:rocm3.9-pytorch . + +FROM rocm/pytorch:rocm3.9_ubuntu18.04_py3.6_pytorch + +RUN apt-get -y install gpg-agent +RUN wget -q -O - http://repo.radeon.com/rocm/rocm.gpg.key | apt-key add - +RUN echo 'deb [arch=amd64] http://repo.radeon.com/rocm/apt/3.9/ xenial main' | tee /etc/apt/sources.list.d/rocm.list + +RUN apt-get -y update +RUN apt-get -y install apt-utils +RUN apt-get -y install build-essential autotools-dev \ + make git curl vim wget rsync jq openssh-server openssh-client sudo \ + iputils-ping net-tools ethtool libcap2 \ + automake autoconf libtool flex doxygen \ + perl lsb-release iproute2 pciutils graphviz \ + bc tar git bash pbzip2 pv bzip2 unzip cabextract \ + g++ gcc \ + && apt-get autoremove + +# sh +RUN rm /bin/sh && ln -s /bin/bash /bin/sh +RUN rm /opt/cache/bin/c++ && \ + rm /opt/cache/bin/cc && \ + rm /opt/cache/bin/g++ && \ + rm /opt/cache/bin/gcc + +# Labels for the docker +LABEL description="This docker sets up the environment to run ORT Training with AMD GPU" + +# CMake +ENV CMAKE_VERSION=3.18.2 +RUN cd /usr/local && \ + wget -q -O - https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-Linux-x86_64.tar.gz | tar zxf - +ENV PATH=/usr/local/cmake-${CMAKE_VERSION}-Linux-x86_64/bin:${PATH} + +ENV WORKSPACE_DIR=/workspace +RUN mkdir -p $WORKSPACE_DIR +WORKDIR $WORKSPACE_DIR + +ENV OLD_PATH=${PATH} +ENV PATH=/usr/bin:${PATH} +# Infiniband setup, openmpi installed under /usr/mpi/gcc/openmpi-4.0.4rc3 doesn't support multi-thread +ENV MOFED_VERSION=5.1-0.6.6.0 +ENV MOFED_OS=ubuntu18.04 +ENV MOFED_FILENAME=MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 +RUN curl -fSsL https://www.mellanox.com/downloads/ofed/MLNX_OFED-${MOFED_VERSION}/${MOFED_FILENAME}.tgz | tar -zxpf - +RUN cd MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 && \ + ./mlnxofedinstall --force --user-space-only --without-fw-update --hpc && \ + cd .. && \ + rm -r MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 + +ENV PATH=${OLD_PATH} +ENV unset OLD_PATH + +# python env +ARG NUMPY_VERSION=1.18.5 +ARG ONNX_VERSION=1.7.0 +RUN pip3 install --no-cache-dir wheel tqdm boto3 requests six ipdb h5py html2text nltk progressbar pyyaml \ + git+https://github.com/NVIDIA/dllogger \ + numpy==${NUMPY_VERSION} \ + onnx=="${ONNX_VERSION}" + +ENV GITHUB_DIR=$WORKSPACE_DIR/github +RUN mkdir -p $GITHUB_DIR + +# UCX +WORKDIR $GITHUB_DIR +RUN apt-get -y update && apt-get -y --no-install-recommends install libnuma-dev +ARG UCX_VERSION=1.9.0-rc3 +ENV UCX_DIR=$WORKSPACE_DIR/ucx-$UCX_VERSION +RUN git clone https://github.com/openucx/ucx.git \ + && cd ucx \ + && git checkout v$UCX_VERSION \ + && ./autogen.sh \ + && mkdir build \ + && cd build \ + && ../contrib/configure-opt --prefix=$UCX_DIR --without-rocm --without-knem --without-cuda \ + && make -j"$(nproc)" \ + && make install + +# OpenMPI +# note: require --enable-orterun-prefix-by-default for Azure machine learning compute +# note: disable verbs as we use ucx middleware and don't want btl openib warnings +WORKDIR $GITHUB_DIR +ARG OPENMPI_BASEVERSION=4.0 +ARG OPENMPI_VERSION=${OPENMPI_BASEVERSION}.5 +ENV OPENMPI_DIR=$WORKSPACE_DIR/openmpi-${OPENMPI_VERSION} +RUN git clone --recursive https://github.com/open-mpi/ompi.git \ + && cd ompi \ + && git checkout v$OPENMPI_VERSION \ + && ./autogen.pl \ + && mkdir build \ + && cd build \ + && ../configure --prefix=$OPENMPI_DIR --with-ucx=$UCX_DIR --without-verbs \ + --enable-mpirun-prefix-by-default --enable-orterun-prefix-by-default \ + --enable-mca-no-build=btl-uct --disable-mpi-fortran \ + && make -j"$(nproc)" \ + && make install \ + && ldconfig \ + && test -f ${OPENMPI_DIR}/bin/mpic++ + +ENV PATH=$OPENMPI_DIR/bin:${PATH} +ENV LD_LIBRARY_PATH=$OPENMPI_DIR/lib:${LD_LIBRARY_PATH} + +# Create a wrapper for OpenMPI to allow running as root by default +RUN mv $OPENMPI_DIR/bin/mpirun $OPENMPI_DIR/bin/mpirun.real && \ + echo '#!/bin/bash' > $OPENMPI_DIR/bin/mpirun && \ + echo 'mpirun.real --allow-run-as-root "$@"' >> $OPENMPI_DIR/bin/mpirun && \ + chmod a+x $OPENMPI_DIR/bin/mpirun + +# install mpi4py (be sure to link existing /opt/openmpi-xxx) +RUN CC=mpicc MPICC=mpicc pip install mpi4py --no-binary mpi4py + +ARG CACHE_DATA=2020-11-09 + +# ONNX Runtime +WORKDIR $GITHUB_DIR +ENV ORT_DIR=$GITHUB_DIR/onnxruntime +RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \ + && cd onnxruntime \ + && python3 tools/ci_build/build.py \ + --cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \ + --build_dir build \ + --config RelWithDebInfo \ + --parallel \ + --skip_tests \ + --build_wheel \ + --use_rocm --rocm_home /opt/rocm \ + --mpi_home $OPENMPI_DIR \ + --nccl_home /opt/rocm \ + --enable_training \ + && test -f $ORT_DIR/build/RelWithDebInfo/onnxruntime_training_bert \ + && pip install $ORT_DIR/build/RelWithDebInfo/dist/*.whl \ + && ldconfig + +# ONNX Runtime Training Examples +WORKDIR $GITHUB_DIR +ARG GPT2_DATASET=wikitext-103 +RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training-examples.git \ + && cd onnxruntime-training-examples \ + # Nvidia BERT + && git clone --no-checkout https://github.com/NVIDIA/DeepLearningExamples.git \ + && cd DeepLearningExamples \ + && git checkout cf54b787 \ + && cd .. \ + && mv DeepLearningExamples/PyTorch/LanguageModeling/BERT ${WORKSPACE_DIR} \ + && rm -rf DeepLearningExamples \ + && cp -r ./nvidia-bert/ort_addon/* ${WORKSPACE_DIR}/BERT \ + # GPT2 fine-tuning + && cd huggingface-gpt2 \ + && git clone https://github.com/huggingface/transformers.git \ + && cd transformers \ + && git checkout 9a0a8c1c6f4f2f0c80ff07d36713a3ada785eec5 \ + && cd .. \ + && mkdir -p ${WORKSPACE_DIR}/GPT2 \ + && cp -r transformers ${WORKSPACE_DIR}/GPT2 \ + && cd ${WORKSPACE_DIR}/GPT2/transformers \ + && git apply $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/src_changes.patch \ + && cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \ + && python3 -m pip install --no-cache-dir -e . \ + && python3 -m pip install --no-cache-dir -r examples/requirements.txt \ + && python3 -m pip install cerberus sympy \ + && cd .. \ + && wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \ + && unzip ${GPT2_DATASET}-v1.zip + +ENV BERT_DIR=${WORKSPACE_DIR}/BERT +ENV GPT2_DIR=${WORKSPACE_DIR}/GPT2 +ENV TRAIN_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.train.tokens +ENV TEST_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.test.tokens + +# Enable ssh access without password needed +RUN sed -i 's/#PermitRootLogin prohibit-password/PermitRootLogin yes/g' /etc/ssh/sshd_config +RUN sed -i 's/#StrictModes yes/StrictModes no/g' /etc/ssh/sshd_config +RUN sed -i 's/#PubkeyAuthentication yes/PubkeyAuthentication yes/g' /etc/ssh/sshd_config +RUN sed -i 's/#PermitEmptyPasswords no/PermitEmptyPasswords yes/g' /etc/ssh/sshd_config + +# Start or Restart sshd service +ENTRYPOINT service ssh restart && /bin/bash + +# Add model and scripts +ADD model ${WORKSPACE_DIR}/model +ADD script ${WORKSPACE_DIR}/script +RUN chmod a+x ${WORKSPACE_DIR}/script/run_bert.sh + +# add locale en_US.UTF-8 +RUN apt-get install -y locales +RUN locale-gen en_US.UTF-8 + +WORKDIR ${WORKSPACE_DIR}/script