mirror of
https://github.com/saymrwulf/onnxruntime.git
synced 2026-07-30 20:18:08 +00:00
revert the code change which was based on b4869926
The change b4869926 which was to remove per-thread allocator would cause seg fault for
distributed training.
In addition, add dockerfile for ROCm3.9
This commit is contained in:
parent
c23fbba463
commit
fc614ad050
7 changed files with 455 additions and 31 deletions
|
|
@ -57,16 +57,30 @@ ONNX_OPERATOR_KERNEL_EX(
|
|||
|
||||
} // namespace rocm
|
||||
|
||||
ROCMExecutionProvider::PerThreadContext::PerThreadContext(OrtDevice::DeviceId device_id) {
|
||||
ROCMExecutionProvider::PerThreadContext::PerThreadContext(OrtDevice::DeviceId device_id, size_t hip_mem_limit, ArenaExtendStrategy arena_extend_strategy) {
|
||||
HIP_CALL_THROW(hipSetDevice(device_id));
|
||||
ROCBLAS_CALL_THROW(rocblas_create_handle(&rocblas_handle_));
|
||||
MIOPEN_CALL_THROW(miopenCreate(&miopen_handle_));
|
||||
|
||||
AllocatorCreationInfo default_memory_info(
|
||||
[](OrtDevice::DeviceId id) {
|
||||
return onnxruntime::make_unique<ROCMAllocator>(id, CUDA);
|
||||
},
|
||||
device_id,
|
||||
true,
|
||||
{hip_mem_limit,
|
||||
static_cast<int>(arena_extend_strategy),
|
||||
-1, -1});
|
||||
|
||||
// HIP malloc/free is expensive so always use an arena
|
||||
allocator_ = CreateAllocator(default_memory_info);
|
||||
}
|
||||
|
||||
ROCMExecutionProvider::PerThreadContext::~PerThreadContext() {
|
||||
// dtor shouldn't throw. if something went wrong earlier (e.g. out of HIP memory) the handles
|
||||
// here may be bad, and the destroy calls can throw.
|
||||
// https://isocpp.github.io/CppCoreGuidelines/CppCoreGuidelines#Rc-dtor-noexcept
|
||||
|
||||
try {
|
||||
ROCBLAS_CALL(rocblas_destroy_handle(rocblas_handle_));
|
||||
} catch (const std::exception& ex) {
|
||||
|
|
@ -200,7 +214,7 @@ ROCMExecutionProvider::PerThreadContext& ROCMExecutionProvider::GetPerThreadCont
|
|||
|
||||
// get or create a context
|
||||
if (context_state_.retired_context_pool.empty()) {
|
||||
context = std::make_shared<PerThreadContext>(device_id_);
|
||||
context = std::make_shared<PerThreadContext>(device_id_, hip_mem_limit_, arena_extend_strategy_);
|
||||
} else {
|
||||
context = context_state_.retired_context_pool.back();
|
||||
context_state_.retired_context_pool.pop_back();
|
||||
|
|
@ -236,6 +250,17 @@ void ROCMExecutionProvider::ReleasePerThreadContext() const {
|
|||
per_thread_context_cache->erase(cached_context_it);
|
||||
}
|
||||
|
||||
AllocatorPtr ROCMExecutionProvider::GetAllocator(int id, OrtMemType mem_type) const {
|
||||
// Pinned memory allocator is shared between threads, but HIP memory allocator is per-thread or it may cause result changes
|
||||
// A hypothesis is that arena allocator is not aligned with HIP output cache, and data from different kernel writes may
|
||||
// cause cacheline to contain dirty data.
|
||||
if (mem_type == OrtMemTypeDefault) {
|
||||
return GetPerThreadContext().GetAllocator();
|
||||
} else {
|
||||
return IExecutionProvider::GetAllocator(id, mem_type);
|
||||
}
|
||||
}
|
||||
|
||||
Status ROCMExecutionProvider::Sync() const {
|
||||
HIP_RETURN_IF_ERROR(hipDeviceSynchronize());
|
||||
return Status::OK();
|
||||
|
|
|
|||
|
|
@ -32,6 +32,8 @@ class ROCMExecutionProvider : public IExecutionProvider {
|
|||
explicit ROCMExecutionProvider(const ROCMExecutionProviderInfo& info);
|
||||
virtual ~ROCMExecutionProvider();
|
||||
|
||||
AllocatorPtr GetAllocator(int id, OrtMemType mem_type) const override;
|
||||
|
||||
Status Sync() const override;
|
||||
|
||||
Status OnRunStart() override;
|
||||
|
|
@ -53,24 +55,7 @@ class ROCMExecutionProvider : public IExecutionProvider {
|
|||
|
||||
template <typename T>
|
||||
const T* GetConstOnes(size_t count) {
|
||||
if (std::is_same<T, float>::value) {
|
||||
if (!constant_ones_float_) {
|
||||
constant_ones_float_ = rocm::CreateConstantOnes<float>();
|
||||
}
|
||||
return reinterpret_cast<const T*>(constant_ones_float_->GetBuffer(count));
|
||||
} else if (std::is_same<T, double>::value) {
|
||||
if (!constant_ones_double_) {
|
||||
constant_ones_double_ = rocm::CreateConstantOnes<double>();
|
||||
}
|
||||
return reinterpret_cast<const T*>(constant_ones_double_->GetBuffer(count));
|
||||
} else if (std::is_same<T, half>::value) {
|
||||
if (!constant_ones_half_) {
|
||||
constant_ones_half_ = rocm::CreateConstantOnes<half>();
|
||||
}
|
||||
return reinterpret_cast<const T*>(constant_ones_half_->GetBuffer(count));
|
||||
} else {
|
||||
return nullptr;
|
||||
}
|
||||
return GetPerThreadContext().template GetConstOnes<T>(count);
|
||||
}
|
||||
|
||||
void AddDeferredReleaseCPUPtr(void* p);
|
||||
|
|
@ -108,13 +93,9 @@ class ROCMExecutionProvider : public IExecutionProvider {
|
|||
std::unordered_map<hipEvent_t, DeferredReleaseCPUPtrs> deferred_release_cpu_ptr_;
|
||||
OrtMutex deferred_release_cpu_ptr_mutex_;
|
||||
|
||||
std::unique_ptr<rocm::IConstantBuffer<float>> constant_ones_float_;
|
||||
std::unique_ptr<rocm::IConstantBuffer<double>> constant_ones_double_;
|
||||
std::unique_ptr<rocm::IConstantBuffer<half>> constant_ones_half_;
|
||||
|
||||
class PerThreadContext final {
|
||||
public:
|
||||
PerThreadContext(OrtDevice::DeviceId device_id);
|
||||
PerThreadContext(OrtDevice::DeviceId device_id, size_t hip_mem_limit, ArenaExtendStrategy arena_extend_strategy);
|
||||
~PerThreadContext();
|
||||
|
||||
rocblas_handle RocblasHandle() const {
|
||||
|
|
@ -129,6 +110,32 @@ class ROCMExecutionProvider : public IExecutionProvider {
|
|||
return current_deferred_release_event_;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
const T* GetConstOnes(size_t count) {
|
||||
if (std::is_same<T, float>::value) {
|
||||
if (!constant_ones_float_) {
|
||||
constant_ones_float_ = rocm::CreateConstantOnes<float>();
|
||||
}
|
||||
return reinterpret_cast<const T*>(constant_ones_float_->GetBuffer(count));
|
||||
} else if (std::is_same<T, double>::value) {
|
||||
if (!constant_ones_double_) {
|
||||
constant_ones_double_ = rocm::CreateConstantOnes<double>();
|
||||
}
|
||||
return reinterpret_cast<const T*>(constant_ones_double_->GetBuffer(count));
|
||||
} else if (std::is_same<T, half>::value) {
|
||||
if (!constant_ones_half_) {
|
||||
constant_ones_half_ = rocm::CreateConstantOnes<half>();
|
||||
}
|
||||
return reinterpret_cast<const T*>(constant_ones_half_->GetBuffer(count));
|
||||
} else {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
AllocatorPtr GetAllocator() const {
|
||||
return allocator_;
|
||||
}
|
||||
|
||||
private:
|
||||
rocblas_handle rocblas_handle_ = nullptr;
|
||||
miopenHandle_t miopen_handle_ = nullptr;
|
||||
|
|
@ -137,6 +144,12 @@ class ROCMExecutionProvider : public IExecutionProvider {
|
|||
// note that hipEvent will be assigned at OnRunEnd() when PerThreadContext destory
|
||||
// so the ownership is passed to deferred_release_cpu_ptr_
|
||||
hipEvent_t current_deferred_release_event_ = nullptr;
|
||||
|
||||
std::unique_ptr<rocm::IConstantBuffer<float>> constant_ones_float_;
|
||||
std::unique_ptr<rocm::IConstantBuffer<double>> constant_ones_double_;
|
||||
std::unique_ptr<rocm::IConstantBuffer<half>> constant_ones_half_;
|
||||
|
||||
AllocatorPtr allocator_;
|
||||
};
|
||||
|
||||
using PerThreadContextMap = std::unordered_map<const ROCMExecutionProvider*, std::weak_ptr<PerThreadContext>>;
|
||||
|
|
|
|||
|
|
@ -120,7 +120,7 @@ RUN CC=mpicc MPICC=mpicc pip install mpi4py --no-binary mpi4py
|
|||
# ONNX Runtime
|
||||
WORKDIR $GITHUB_DIR
|
||||
ENV ORT_DIR=$GITHUB_DIR/onnxruntime
|
||||
RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxruntime.git \
|
||||
RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \
|
||||
&& cd onnxruntime \
|
||||
&& python3 tools/ci_build/build.py \
|
||||
--cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \
|
||||
|
|
|
|||
|
|
@ -114,7 +114,7 @@ ARG CACHE_DATA=2020-10-28
|
|||
# ONNX Runtime
|
||||
WORKDIR $GITHUB_DIR
|
||||
ENV ORT_DIR=$GITHUB_DIR/onnxruntime
|
||||
RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxruntime.git \
|
||||
RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \
|
||||
&& cd onnxruntime \
|
||||
&& python3 tools/ci_build/build.py \
|
||||
--cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \
|
||||
|
|
@ -126,7 +126,7 @@ RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxrunt
|
|||
--use_rocm --rocm_home /opt/rocm \
|
||||
--mpi_home $OPENMPI_DIR \
|
||||
--nccl_home /opt/rocm \
|
||||
--enable_training \
|
||||
--enable_training \
|
||||
&& test -f $ORT_DIR/build/RelWithDebInfo/onnxruntime_training_bert \
|
||||
&& pip install $ORT_DIR/build/RelWithDebInfo/dist/*.whl \
|
||||
&& ldconfig
|
||||
|
|
@ -157,7 +157,7 @@ RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training
|
|||
&& cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \
|
||||
&& python3 -m pip install --no-cache-dir -e . \
|
||||
&& python3 -m pip install --no-cache-dir -r examples/requirements.txt \
|
||||
&& python3 -m pip install cerberus \
|
||||
&& python3 -m pip install cerberus sympy \
|
||||
&& cd .. \
|
||||
&& wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \
|
||||
&& unzip ${GPT2_DATASET}-v1.zip
|
||||
|
|
|
|||
|
|
@ -113,7 +113,7 @@ ARG CACHE_DATA=2020-10-28
|
|||
# ONNX Runtime
|
||||
WORKDIR $GITHUB_DIR
|
||||
ENV ORT_DIR=$GITHUB_DIR/onnxruntime
|
||||
RUN git clone --recursive -b wezhan/amdgpu https://github.com/microsoft/onnxruntime.git \
|
||||
RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \
|
||||
&& cd onnxruntime \
|
||||
&& python3 tools/ci_build/build.py \
|
||||
--cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \
|
||||
|
|
@ -156,7 +156,7 @@ RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training
|
|||
&& cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \
|
||||
&& python3 -m pip install --no-cache-dir -e . \
|
||||
&& python3 -m pip install --no-cache-dir -r examples/requirements.txt \
|
||||
&& python3 -m pip install cerberus \
|
||||
&& python3 -m pip install cerberus sympy \
|
||||
&& cd .. \
|
||||
&& wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \
|
||||
&& unzip ${GPT2_DATASET}-v1.zip
|
||||
|
|
|
|||
196
orttraining/tools/amdgpu/Dockerfile.rocm3.9
Normal file
196
orttraining/tools/amdgpu/Dockerfile.rocm3.9
Normal file
|
|
@ -0,0 +1,196 @@
|
|||
# docker build --network=host --file Dockerfile.rocm3.9 --tag ort:rocm3.9-ort-dev .
|
||||
|
||||
FROM rocm/tensorflow:rocm3.9-tf2.3-dev
|
||||
|
||||
RUN wget -q -O - http://repo.radeon.com/rocm/rocm.gpg.key | apt-key add -
|
||||
RUN echo 'deb [arch=amd64] http://repo.radeon.com/rocm/apt/3.9/ xenial main' | tee /etc/apt/sources.list.d/rocm.list
|
||||
|
||||
RUN apt-get -y update
|
||||
RUN apt-get -y install apt-utils
|
||||
RUN apt-get -y install build-essential autotools-dev \
|
||||
make git curl vim wget rsync jq openssh-server openssh-client sudo \
|
||||
iputils-ping net-tools ethtool libcap2 \
|
||||
automake autoconf libtool flex doxygen \
|
||||
perl lsb-release iproute2 pciutils graphviz \
|
||||
bc tar git bash pbzip2 pv bzip2 cabextract \
|
||||
g++ gcc \
|
||||
&& apt-get autoremove
|
||||
|
||||
# sh
|
||||
RUN rm /bin/sh && ln -s /bin/bash /bin/sh
|
||||
|
||||
# Labels for the docker
|
||||
LABEL description="This docker sets up the environment to run ORT Training with AMD GPU"
|
||||
|
||||
# CMake
|
||||
ENV CMAKE_VERSION=3.18.2
|
||||
RUN cd /usr/local && \
|
||||
wget -q -O - https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-Linux-x86_64.tar.gz | tar zxf -
|
||||
ENV PATH=/usr/local/cmake-${CMAKE_VERSION}-Linux-x86_64/bin:${PATH}
|
||||
|
||||
# WORKSPACE_DIR
|
||||
ENV WORKSPACE_DIR=/workspace
|
||||
RUN mkdir -p $WORKSPACE_DIR
|
||||
WORKDIR $WORKSPACE_DIR
|
||||
|
||||
# Infiniband setup, openmpi installed under /usr/mpi/gcc/openmpi-4.0.4rc3 doesn't support multi-thread
|
||||
ENV MOFED_VERSION=5.1-0.6.6.0
|
||||
ENV MOFED_OS=ubuntu18.04
|
||||
ENV MOFED_FILENAME=MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64
|
||||
RUN curl -fSsL https://www.mellanox.com/downloads/ofed/MLNX_OFED-${MOFED_VERSION}/${MOFED_FILENAME}.tgz | tar -zxpf -
|
||||
RUN cd MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 && \
|
||||
./mlnxofedinstall --force --user-space-only --without-fw-update --hpc && \
|
||||
cd .. && \
|
||||
rm -r MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64
|
||||
|
||||
# install miniconda (comes with python 3.9 default)
|
||||
ARG CONDA_VERSION=4.7.10
|
||||
ARG CONDA_URL=https://repo.anaconda.com/miniconda/Miniconda3-${CONDA_VERSION}-Linux-x86_64.sh
|
||||
RUN curl -fSsL --insecure ${CONDA_URL} -o install-conda.sh &&\
|
||||
/bin/bash ./install-conda.sh -b -p /opt/conda &&\
|
||||
/opt/conda/bin/conda clean -ya
|
||||
ENV PATH=/opt/conda/bin:${PATH}
|
||||
|
||||
ARG NUMPY_VERSION=1.18.5
|
||||
ARG ONNX_VERSION=1.7.0
|
||||
RUN conda install -y \
|
||||
numpy=${NUMPY_VERSION} \
|
||||
cmake \
|
||||
ninja \
|
||||
pyyaml \
|
||||
cffi \
|
||||
setuptools \
|
||||
&& pip install --no-cache-dir wheel tqdm boto3 requests six ipdb h5py html2text nltk progressbar \
|
||||
git+https://github.com/NVIDIA/dllogger \
|
||||
onnx=="${ONNX_VERSION}"
|
||||
|
||||
# GITHUB_DIR
|
||||
ENV GITHUB_DIR=$WORKSPACE_DIR/github
|
||||
RUN mkdir -p $GITHUB_DIR
|
||||
|
||||
# UCX
|
||||
WORKDIR $GITHUB_DIR
|
||||
RUN apt-get -y update && apt-get -y --no-install-recommends install libnuma-dev
|
||||
ARG UCX_VERSION=1.9.0-rc3
|
||||
ENV UCX_DIR=$WORKSPACE_DIR/ucx-$UCX_VERSION
|
||||
RUN git clone https://github.com/openucx/ucx.git \
|
||||
&& cd ucx \
|
||||
&& git checkout v$UCX_VERSION \
|
||||
&& ./autogen.sh \
|
||||
&& mkdir build \
|
||||
&& cd build \
|
||||
&& ../contrib/configure-opt --prefix=$UCX_DIR --without-rocm --without-knem --without-cuda \
|
||||
&& make -j"$(nproc)" \
|
||||
&& make install
|
||||
|
||||
# OpenMPI
|
||||
# note: require --enable-orterun-prefix-by-default for Azure machine learning compute
|
||||
# note: disable verbs as we use ucx middleware and don't want btl openib warnings
|
||||
WORKDIR $GITHUB_DIR
|
||||
ARG OPENMPI_BASEVERSION=4.0
|
||||
ARG OPENMPI_VERSION=${OPENMPI_BASEVERSION}.5
|
||||
ENV OPENMPI_DIR=$WORKSPACE_DIR/openmpi-${OPENMPI_VERSION}
|
||||
RUN git clone --recursive https://github.com/open-mpi/ompi.git \
|
||||
&& cd ompi \
|
||||
&& git checkout v$OPENMPI_VERSION \
|
||||
&& ./autogen.pl \
|
||||
&& mkdir build \
|
||||
&& cd build \
|
||||
&& ../configure --prefix=$OPENMPI_DIR --with-ucx=$UCX_DIR --without-verbs \
|
||||
--enable-mpirun-prefix-by-default --enable-orterun-prefix-by-default \
|
||||
--enable-mca-no-build=btl-uct --disable-mpi-fortran \
|
||||
&& make -j"$(nproc)" \
|
||||
&& make install \
|
||||
&& ldconfig \
|
||||
&& test -f ${OPENMPI_DIR}/bin/mpic++
|
||||
|
||||
ENV PATH=$OPENMPI_DIR/bin:${PATH}
|
||||
ENV LD_LIBRARY_PATH=$OPENMPI_DIR/lib:${LD_LIBRARY_PATH}
|
||||
|
||||
# Create a wrapper for OpenMPI to allow running as root by default
|
||||
RUN mv $OPENMPI_DIR/bin/mpirun $OPENMPI_DIR/bin/mpirun.real && \
|
||||
echo '#!/bin/bash' > $OPENMPI_DIR/bin/mpirun && \
|
||||
echo 'mpirun.real --allow-run-as-root "$@"' >> $OPENMPI_DIR/bin/mpirun && \
|
||||
chmod a+x $OPENMPI_DIR/bin/mpirun
|
||||
|
||||
# install mpi4py (be sure to link existing /opt/openmpi-xxx)
|
||||
RUN CC=mpicc MPICC=mpicc pip install mpi4py --no-binary mpi4py
|
||||
|
||||
# ONNX Runtime
|
||||
WORKDIR $GITHUB_DIR
|
||||
ENV ORT_DIR=$GITHUB_DIR/onnxruntime
|
||||
RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \
|
||||
&& cd onnxruntime \
|
||||
&& python3 tools/ci_build/build.py \
|
||||
--cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \
|
||||
--build_dir build \
|
||||
--config RelWithDebInfo \
|
||||
--parallel \
|
||||
--skip_tests \
|
||||
--build_wheel \
|
||||
--use_rocm --rocm_home /opt/rocm \
|
||||
--mpi_home $OPENMPI_DIR \
|
||||
--nccl_home /opt/rocm \
|
||||
--enable_training \
|
||||
&& test -f $ORT_DIR/build/RelWithDebInfo/onnxruntime_training_bert \
|
||||
&& pip install $ORT_DIR/build/RelWithDebInfo/dist/*.whl \
|
||||
&& ldconfig
|
||||
|
||||
# Instructions to pull and install the nightly ROCm3.8 PyTorch whl pacakge
|
||||
RUN pip3 install --pre torch -f https://download.pytorch.org/whl/nightly/rocm3.9/torch_nightly.html
|
||||
|
||||
# ONNX Runtime Training Examples
|
||||
WORKDIR $GITHUB_DIR
|
||||
ARG GPT2_DATASET=wikitext-103
|
||||
RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training-examples.git \
|
||||
&& cd onnxruntime-training-examples \
|
||||
# Nvidia BERT
|
||||
&& git clone --no-checkout https://github.com/NVIDIA/DeepLearningExamples.git \
|
||||
&& cd DeepLearningExamples \
|
||||
&& git checkout cf54b787 \
|
||||
&& cd .. \
|
||||
&& mv DeepLearningExamples/PyTorch/LanguageModeling/BERT ${WORKSPACE_DIR} \
|
||||
&& rm -rf DeepLearningExamples \
|
||||
&& cp -r ./nvidia-bert/ort_addon/* ${WORKSPACE_DIR}/BERT \
|
||||
# GPT2 fine-tuning
|
||||
&& cd huggingface-gpt2 \
|
||||
&& git clone https://github.com/huggingface/transformers.git \
|
||||
&& cd transformers \
|
||||
&& git checkout 9a0a8c1c6f4f2f0c80ff07d36713a3ada785eec5 \
|
||||
&& cd .. \
|
||||
&& mkdir -p ${WORKSPACE_DIR}/GPT2 \
|
||||
&& cp -r transformers ${WORKSPACE_DIR}/GPT2 \
|
||||
&& cd ${WORKSPACE_DIR}/GPT2/transformers \
|
||||
&& git apply $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/src_changes.patch \
|
||||
&& cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \
|
||||
&& python3 -m pip install --no-cache-dir -e . \
|
||||
&& python3 -m pip install --no-cache-dir -r examples/requirements.txt \
|
||||
&& python3 -m pip install cerberus sympy packaging \
|
||||
&& cd .. \
|
||||
&& wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \
|
||||
&& unzip ${GPT2_DATASET}-v1.zip
|
||||
|
||||
ENV BERT_DIR=${WORKSPACE_DIR}/BERT
|
||||
ENV GPT2_DIR=${WORKSPACE_DIR}/GPT2
|
||||
ENV TRAIN_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.train.tokens
|
||||
ENV TEST_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.test.tokens
|
||||
|
||||
# Enable ssh access without password needed
|
||||
RUN sed -i 's/#PermitRootLogin prohibit-password/PermitRootLogin yes/g' /etc/ssh/sshd_config
|
||||
RUN sed -i 's/#StrictModes yes/StrictModes no/g' /etc/ssh/sshd_config
|
||||
RUN sed -i 's/#PubkeyAuthentication yes/PubkeyAuthentication yes/g' /etc/ssh/sshd_config
|
||||
RUN sed -i 's/#PermitEmptyPasswords no/PermitEmptyPasswords yes/g' /etc/ssh/sshd_config
|
||||
|
||||
# Start or Restart sshd service
|
||||
ENTRYPOINT service ssh restart && /bin/bash
|
||||
|
||||
# Add model and scripts
|
||||
ADD model ${WORKSPACE_DIR}/model
|
||||
ADD script ${WORKSPACE_DIR}/script
|
||||
RUN chmod a+x ${WORKSPACE_DIR}/script/run_bert.sh
|
||||
|
||||
# add locale en_US.UTF-8
|
||||
RUN apt-get install -y locales
|
||||
RUN locale-gen en_US.UTF-8
|
||||
|
||||
WORKDIR ${WORKSPACE_DIR}/script
|
||||
190
orttraining/tools/amdgpu/Dockerfile.rocm3.9.pytorch
Normal file
190
orttraining/tools/amdgpu/Dockerfile.rocm3.9.pytorch
Normal file
|
|
@ -0,0 +1,190 @@
|
|||
# docker build --network=host --file Dockerfile.rocm3.9.pytorch --tag ort:rocm3.9-pytorch .
|
||||
|
||||
FROM rocm/pytorch:rocm3.9_ubuntu18.04_py3.6_pytorch
|
||||
|
||||
RUN apt-get -y install gpg-agent
|
||||
RUN wget -q -O - http://repo.radeon.com/rocm/rocm.gpg.key | apt-key add -
|
||||
RUN echo 'deb [arch=amd64] http://repo.radeon.com/rocm/apt/3.9/ xenial main' | tee /etc/apt/sources.list.d/rocm.list
|
||||
|
||||
RUN apt-get -y update
|
||||
RUN apt-get -y install apt-utils
|
||||
RUN apt-get -y install build-essential autotools-dev \
|
||||
make git curl vim wget rsync jq openssh-server openssh-client sudo \
|
||||
iputils-ping net-tools ethtool libcap2 \
|
||||
automake autoconf libtool flex doxygen \
|
||||
perl lsb-release iproute2 pciutils graphviz \
|
||||
bc tar git bash pbzip2 pv bzip2 unzip cabextract \
|
||||
g++ gcc \
|
||||
&& apt-get autoremove
|
||||
|
||||
# sh
|
||||
RUN rm /bin/sh && ln -s /bin/bash /bin/sh
|
||||
RUN rm /opt/cache/bin/c++ && \
|
||||
rm /opt/cache/bin/cc && \
|
||||
rm /opt/cache/bin/g++ && \
|
||||
rm /opt/cache/bin/gcc
|
||||
|
||||
# Labels for the docker
|
||||
LABEL description="This docker sets up the environment to run ORT Training with AMD GPU"
|
||||
|
||||
# CMake
|
||||
ENV CMAKE_VERSION=3.18.2
|
||||
RUN cd /usr/local && \
|
||||
wget -q -O - https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-Linux-x86_64.tar.gz | tar zxf -
|
||||
ENV PATH=/usr/local/cmake-${CMAKE_VERSION}-Linux-x86_64/bin:${PATH}
|
||||
|
||||
ENV WORKSPACE_DIR=/workspace
|
||||
RUN mkdir -p $WORKSPACE_DIR
|
||||
WORKDIR $WORKSPACE_DIR
|
||||
|
||||
ENV OLD_PATH=${PATH}
|
||||
ENV PATH=/usr/bin:${PATH}
|
||||
# Infiniband setup, openmpi installed under /usr/mpi/gcc/openmpi-4.0.4rc3 doesn't support multi-thread
|
||||
ENV MOFED_VERSION=5.1-0.6.6.0
|
||||
ENV MOFED_OS=ubuntu18.04
|
||||
ENV MOFED_FILENAME=MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64
|
||||
RUN curl -fSsL https://www.mellanox.com/downloads/ofed/MLNX_OFED-${MOFED_VERSION}/${MOFED_FILENAME}.tgz | tar -zxpf -
|
||||
RUN cd MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64 && \
|
||||
./mlnxofedinstall --force --user-space-only --without-fw-update --hpc && \
|
||||
cd .. && \
|
||||
rm -r MLNX_OFED_LINUX-${MOFED_VERSION}-${MOFED_OS}-x86_64
|
||||
|
||||
ENV PATH=${OLD_PATH}
|
||||
ENV unset OLD_PATH
|
||||
|
||||
# python env
|
||||
ARG NUMPY_VERSION=1.18.5
|
||||
ARG ONNX_VERSION=1.7.0
|
||||
RUN pip3 install --no-cache-dir wheel tqdm boto3 requests six ipdb h5py html2text nltk progressbar pyyaml \
|
||||
git+https://github.com/NVIDIA/dllogger \
|
||||
numpy==${NUMPY_VERSION} \
|
||||
onnx=="${ONNX_VERSION}"
|
||||
|
||||
ENV GITHUB_DIR=$WORKSPACE_DIR/github
|
||||
RUN mkdir -p $GITHUB_DIR
|
||||
|
||||
# UCX
|
||||
WORKDIR $GITHUB_DIR
|
||||
RUN apt-get -y update && apt-get -y --no-install-recommends install libnuma-dev
|
||||
ARG UCX_VERSION=1.9.0-rc3
|
||||
ENV UCX_DIR=$WORKSPACE_DIR/ucx-$UCX_VERSION
|
||||
RUN git clone https://github.com/openucx/ucx.git \
|
||||
&& cd ucx \
|
||||
&& git checkout v$UCX_VERSION \
|
||||
&& ./autogen.sh \
|
||||
&& mkdir build \
|
||||
&& cd build \
|
||||
&& ../contrib/configure-opt --prefix=$UCX_DIR --without-rocm --without-knem --without-cuda \
|
||||
&& make -j"$(nproc)" \
|
||||
&& make install
|
||||
|
||||
# OpenMPI
|
||||
# note: require --enable-orterun-prefix-by-default for Azure machine learning compute
|
||||
# note: disable verbs as we use ucx middleware and don't want btl openib warnings
|
||||
WORKDIR $GITHUB_DIR
|
||||
ARG OPENMPI_BASEVERSION=4.0
|
||||
ARG OPENMPI_VERSION=${OPENMPI_BASEVERSION}.5
|
||||
ENV OPENMPI_DIR=$WORKSPACE_DIR/openmpi-${OPENMPI_VERSION}
|
||||
RUN git clone --recursive https://github.com/open-mpi/ompi.git \
|
||||
&& cd ompi \
|
||||
&& git checkout v$OPENMPI_VERSION \
|
||||
&& ./autogen.pl \
|
||||
&& mkdir build \
|
||||
&& cd build \
|
||||
&& ../configure --prefix=$OPENMPI_DIR --with-ucx=$UCX_DIR --without-verbs \
|
||||
--enable-mpirun-prefix-by-default --enable-orterun-prefix-by-default \
|
||||
--enable-mca-no-build=btl-uct --disable-mpi-fortran \
|
||||
&& make -j"$(nproc)" \
|
||||
&& make install \
|
||||
&& ldconfig \
|
||||
&& test -f ${OPENMPI_DIR}/bin/mpic++
|
||||
|
||||
ENV PATH=$OPENMPI_DIR/bin:${PATH}
|
||||
ENV LD_LIBRARY_PATH=$OPENMPI_DIR/lib:${LD_LIBRARY_PATH}
|
||||
|
||||
# Create a wrapper for OpenMPI to allow running as root by default
|
||||
RUN mv $OPENMPI_DIR/bin/mpirun $OPENMPI_DIR/bin/mpirun.real && \
|
||||
echo '#!/bin/bash' > $OPENMPI_DIR/bin/mpirun && \
|
||||
echo 'mpirun.real --allow-run-as-root "$@"' >> $OPENMPI_DIR/bin/mpirun && \
|
||||
chmod a+x $OPENMPI_DIR/bin/mpirun
|
||||
|
||||
# install mpi4py (be sure to link existing /opt/openmpi-xxx)
|
||||
RUN CC=mpicc MPICC=mpicc pip install mpi4py --no-binary mpi4py
|
||||
|
||||
ARG CACHE_DATA=2020-11-09
|
||||
|
||||
# ONNX Runtime
|
||||
WORKDIR $GITHUB_DIR
|
||||
ENV ORT_DIR=$GITHUB_DIR/onnxruntime
|
||||
RUN git clone --recursive https://github.com/microsoft/onnxruntime.git \
|
||||
&& cd onnxruntime \
|
||||
&& python3 tools/ci_build/build.py \
|
||||
--cmake_extra_defines ONNXRUNTIME_VERSION=`cat ./VERSION_NUMBER` \
|
||||
--build_dir build \
|
||||
--config RelWithDebInfo \
|
||||
--parallel \
|
||||
--skip_tests \
|
||||
--build_wheel \
|
||||
--use_rocm --rocm_home /opt/rocm \
|
||||
--mpi_home $OPENMPI_DIR \
|
||||
--nccl_home /opt/rocm \
|
||||
--enable_training \
|
||||
&& test -f $ORT_DIR/build/RelWithDebInfo/onnxruntime_training_bert \
|
||||
&& pip install $ORT_DIR/build/RelWithDebInfo/dist/*.whl \
|
||||
&& ldconfig
|
||||
|
||||
# ONNX Runtime Training Examples
|
||||
WORKDIR $GITHUB_DIR
|
||||
ARG GPT2_DATASET=wikitext-103
|
||||
RUN git clone -b wezhan/amdgpu https://github.com/microsoft/onnxruntime-training-examples.git \
|
||||
&& cd onnxruntime-training-examples \
|
||||
# Nvidia BERT
|
||||
&& git clone --no-checkout https://github.com/NVIDIA/DeepLearningExamples.git \
|
||||
&& cd DeepLearningExamples \
|
||||
&& git checkout cf54b787 \
|
||||
&& cd .. \
|
||||
&& mv DeepLearningExamples/PyTorch/LanguageModeling/BERT ${WORKSPACE_DIR} \
|
||||
&& rm -rf DeepLearningExamples \
|
||||
&& cp -r ./nvidia-bert/ort_addon/* ${WORKSPACE_DIR}/BERT \
|
||||
# GPT2 fine-tuning
|
||||
&& cd huggingface-gpt2 \
|
||||
&& git clone https://github.com/huggingface/transformers.git \
|
||||
&& cd transformers \
|
||||
&& git checkout 9a0a8c1c6f4f2f0c80ff07d36713a3ada785eec5 \
|
||||
&& cd .. \
|
||||
&& mkdir -p ${WORKSPACE_DIR}/GPT2 \
|
||||
&& cp -r transformers ${WORKSPACE_DIR}/GPT2 \
|
||||
&& cd ${WORKSPACE_DIR}/GPT2/transformers \
|
||||
&& git apply $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/src_changes.patch \
|
||||
&& cp -r $GITHUB_DIR/onnxruntime-training-examples/huggingface-gpt2/ort_addon/ort_supplement/* ./ \
|
||||
&& python3 -m pip install --no-cache-dir -e . \
|
||||
&& python3 -m pip install --no-cache-dir -r examples/requirements.txt \
|
||||
&& python3 -m pip install cerberus sympy \
|
||||
&& cd .. \
|
||||
&& wget https://s3.amazonaws.com/research.metamind.io/wikitext/${GPT2_DATASET}-v1.zip \
|
||||
&& unzip ${GPT2_DATASET}-v1.zip
|
||||
|
||||
ENV BERT_DIR=${WORKSPACE_DIR}/BERT
|
||||
ENV GPT2_DIR=${WORKSPACE_DIR}/GPT2
|
||||
ENV TRAIN_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.train.tokens
|
||||
ENV TEST_FILE=${WORKSPACE_DIR}/GPT2/${GPT2_DATASET}/wiki.test.tokens
|
||||
|
||||
# Enable ssh access without password needed
|
||||
RUN sed -i 's/#PermitRootLogin prohibit-password/PermitRootLogin yes/g' /etc/ssh/sshd_config
|
||||
RUN sed -i 's/#StrictModes yes/StrictModes no/g' /etc/ssh/sshd_config
|
||||
RUN sed -i 's/#PubkeyAuthentication yes/PubkeyAuthentication yes/g' /etc/ssh/sshd_config
|
||||
RUN sed -i 's/#PermitEmptyPasswords no/PermitEmptyPasswords yes/g' /etc/ssh/sshd_config
|
||||
|
||||
# Start or Restart sshd service
|
||||
ENTRYPOINT service ssh restart && /bin/bash
|
||||
|
||||
# Add model and scripts
|
||||
ADD model ${WORKSPACE_DIR}/model
|
||||
ADD script ${WORKSPACE_DIR}/script
|
||||
RUN chmod a+x ${WORKSPACE_DIR}/script/run_bert.sh
|
||||
|
||||
# add locale en_US.UTF-8
|
||||
RUN apt-get install -y locales
|
||||
RUN locale-gen en_US.UTF-8
|
||||
|
||||
WORKDIR ${WORKSPACE_DIR}/script
|
||||
Loading…
Reference in a new issue