-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathContainerFile
More file actions
161 lines (134 loc) · 4.87 KB
/
Copy pathContainerFile
File metadata and controls
161 lines (134 loc) · 4.87 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
#Set Versions.
ARG FEDORA_VERSION=41
ARG OS_TYPE=x86_64
ARG GCC_VERSION=13
#Base image
FROM quay.io/foundata/fedora${FEDORA_VERSION}-itt:latest
USER root
# Use the above args
ARG OS_TYPE
ARG FEDORA_VERSION
ARG GCC_VERSION
ARG PY_VER
ENV HOME=/root
ENV GCC_VERSION=${GCC_VERSION}
ENV CCACHE_DISABLE=1
RUN dnf upgrade --refresh -y && \
dnf install -y \
curl \
wget \
jq \
vim \
git \
gdb \
unzip \
which \
libffi-devel \
openssl-devel \
make \
cmake \
ninja-build \
ccache \
uv \
zsh \
findutils \
python${PY_VER} \
python${PY_VER}-devel \
python${PY_VER}-test && \
dnf clean all
RUN chsh -s $(which zsh)
# Copy the bash script into the image
COPY ./.gitlab/ci/scripts/tools/install_gcc.sh /usr/local/bin/install_gcc_version.sh
RUN bash /usr/local/bin/install_gcc_version.sh
ENV CC=/usr/bin/gcc-13 \
CXX=/usr/bin/g++-13 \
CUDAHOSTCXX=/usr/bin/g++-13
RUN mkdir ./cuda-toolkit && \
cd ./cuda-toolkit && \
wget https://developer.download.nvidia.com/compute/cuda/12.8.1/local_installers/cuda-repo-fedora41-12-8-local-12.8.1_570.124.06-1.x86_64.rpm && \
rpm -i cuda-repo-fedora41-12-8-local-12.8.1_570.124.06-1.x86_64.rpm && \
dnf clean all && \
dnf -y install cuda-toolkit-12-8 && \
cd .. && \
rm -rf ./cuda-toolkit
# CUDA env
ENV CUDA_HOME=/usr/local/cuda
ENV PATH="/usr/local/cuda-12.8/bin:${PATH}"
ENV LD_LIBRARY_PATH="/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
# NVCC host compiler path
ENV NVCC_CCBIN=/usr/bin/g++
RUN echo 'export NVCC_CCBIN="$(command -v g++)"' > /etc/profile.d/nvcc_ccbin.sh
RUN dnf remove -y xz \
&& cd /tmp \
&& curl -LO https://tukaani.org/xz/xz-5.2.5.tar.gz \
&& tar -xzf xz-5.2.5.tar.gz \
&& cd xz-5.2.5 \
&& ./configure --disable-sandbox \
&& make -j"$(nproc)" \
&& make install \
&& cd /tmp \
&& rm -rf xz-5.2.5* \
&& dnf clean all \
&& rm -rf /var/cache/dnf
#Install cudnn
RUN mkdir ./cudnn
RUN cd ./cudnn \
&& wget https://developer.download.nvidia.com/compute/cudnn/redist/cudnn/linux-x86_64/cudnn-linux-x86_64-9.6.0.74_cuda12-archive.tar.xz \
&& tar xvf cudnn-linux-x86_64-9.6.0.74_cuda12-archive.tar.xz \
&& mv ./cudnn-linux-x86_64-9.6.0.74_cuda12-archive/include/* /usr/local/cuda/include \
&& mv ./cudnn-linux-x86_64-9.6.0.74_cuda12-archive/lib/* /usr/local/cuda/lib64 \
&& cd .. && rm -r ./cudnn
#Update file permissions
RUN chmod a+r /usr/local/cuda/include/cudnn*.h /usr/local/cuda/lib64/libcudnn*
# Create dev user before setting up virtualenv
RUN dnf install -y sudo shadow-utils && dnf clean all
# Fix sudo permissions (setuid bit required for non-root users)
RUN chown root:root /usr/bin/sudo && chmod 4755 /usr/bin/sudo
ARG USERNAME=dev
ARG USER_UID=1011
ARG USER_GID=$USER_UID
RUN groupadd --gid $USER_GID $USERNAME && \
useradd --uid $USER_UID --gid $USER_GID -m -s /bin/bash $USERNAME && \
echo "$USERNAME ALL=(ALL) NOPASSWD:ALL" >> /etc/sudoers.d/$USERNAME && \
chmod 0440 /etc/sudoers.d/$USERNAME
# Switch HOME to dev user's home before creating virtualenv
ENV HOME=/home/$USERNAME
# Install Python with uv and setup virtualenv
RUN mkdir -p /pytorch/
COPY ./pytorch/pyproject.toml /pytorch/
RUN cd $HOME && \
uv python install "${PY_VER}" && \
uv venv --python "${PY_VER}" --clear $HOME/.venv && \
bash -c "source $HOME/.venv/bin/activate && uv pip install --upgrade pip" && \
bash -c "source $HOME/.venv/bin/activate && cd /pytorch && uv pip install --group dev && uv pip install mkl-static mkl-include"
#Install for magma
RUN mkdir -p /pytorch/.ci/docker/common/
COPY ./pytorch/.ci/docker/common/install_magma.sh /pytorch/.ci/docker/common/
RUN bash -c "source $HOME/.venv/bin/activate && cd /pytorch && bash .ci/docker/common/install_magma.sh 12.8"
#Install triton.
RUN mkdir -p /pytorch/scripts
COPY ./pytorch/scripts/install_triton_wheel.sh /pytorch/scripts
RUN mkdir -p /pytorch/.ci/docker/
COPY ./pytorch/.ci/docker/triton_version.txt /pytorch/.ci/docker/triton_xpu_version.txt /pytorch/.ci/docker/
RUN mkdir -p /pytorch/.ci/docker/ci_commit_pins/
COPY ./pytorch/.ci/docker/ci_commit_pins/triton.txt /pytorch/.ci/docker/ci_commit_pins/triton-xpu.txt /pytorch/.ci/docker/ci_commit_pins/
COPY ./pytorch/Makefile /pytorch/
RUN bash -c "source $HOME/.venv/bin/activate && cd /pytorch && make triton"
RUN ldconfig
RUN echo "=== Compiler Verification ===" && \
gcc --version && \
g++ --version && \
nvcc --version && \
echo "=== C++ Header Directories ===" && \
ls -la /usr/include/c++/ && \
echo "=== Environment Variables ===" && \
echo "CC=$CC" && \
echo "CXX=$CXX" && \
echo "NVCC_CCBIN=$NVCC_CCBIN" && \
echo "CPLUS_INCLUDE_PATH=$CPLUS_INCLUDE_PATH"
# Fix ownership of venv created as root
RUN chown -R $USERNAME:$USERNAME $HOME/.venv
# Before switching to dev user
RUN mkdir -p /home/dev && \
chown -R $USERNAME:$USERNAME /home/dev
USER $USERNAME