Spaces:

Dovakiins
/

qwerrwe

Build error

App Files Files Community

winglian commited on Jul 26, 2023

Commit

2c37bf6

unverified ·

1 Parent(s): 9f69c4d

Prune cuda117 (#327)

Browse files

* drop cuda117/torch 1.13.1 from support, pin flash attention to v2.0.1, rm torchvision/torchaudio install

* gptq base build not needed. add sm 9.0 support

Files changed (3) hide show

.github/workflows/base.yml +3 -13
.github/workflows/main.yml +1 -11
docker/Dockerfile-base +8 -5

.github/workflows/base.yml CHANGED Viewed

@@ -19,22 +19,12 @@ jobs:
             cuda_version: 11.8.0
             python_version: "3.9"
             pytorch: 2.0.1
-            axolotl_extras:
           - cuda: "118"
             cuda_version: 11.8.0
             python_version: "3.10"
             pytorch: 2.0.1
-            axolotl_extras:
-          - cuda: "117"
-            cuda_version: 11.7.1
-            python_version: "3.9"
-            pytorch: 1.13.1
-            axolotl_extras:
-          - cuda: "118"
-            cuda_version: 11.8.0
-            python_version: "3.9"
-            pytorch: 2.0.1
-            axolotl_extras: gptq
     steps:
       - name: Checkout
         uses: actions/checkout@v3
@@ -63,4 +53,4 @@ jobs:
             CUDA=${{ matrix.cuda }}
             PYTHON_VERSION=${{ matrix.python_version }}
             PYTORCH_VERSION=${{ matrix.pytorch }}
-            AXOLOTL_EXTRAS=${{ matrix.axolotl_extras }}

             cuda_version: 11.8.0
             python_version: "3.9"
             pytorch: 2.0.1
+            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 9.0+PTX"
           - cuda: "118"
             cuda_version: 11.8.0
             python_version: "3.10"
             pytorch: 2.0.1
+            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 9.0+PTX"
     steps:
       - name: Checkout
         uses: actions/checkout@v3
             CUDA=${{ matrix.cuda }}
             PYTHON_VERSION=${{ matrix.python_version }}
             PYTORCH_VERSION=${{ matrix.pytorch }}
+            TORCH_CUDA_ARCH_LIST=${{ matrix.torch_cuda_arch_list }}

.github/workflows/main.yml CHANGED Viewed

@@ -29,11 +29,6 @@ jobs:
             python_version: "3.9"
             pytorch: 2.0.1
             axolotl_extras: gptq
-          - cuda: cu117
-            cuda_version: 11.7.1
-            python_version: "3.9"
-            pytorch: 1.13.1
-            axolotl_extras:
     runs-on: self-hosted
     steps:
       - name: Checkout
@@ -55,7 +50,7 @@ jobs:
         with:
           context: .
           build-args: |
-            BASE_TAG=${{ github.ref_name }}-base-py${{ matrix.python_version }}-${{ matrix.cuda }}-${{ matrix.pytorch }}${{ matrix.axolotl_extras != '' && '-' || '' }}${{ matrix.axolotl_extras }}
           file: ./docker/Dockerfile
           push: ${{ github.event_name != 'pull_request' }}
           tags: ${{ steps.metadata.outputs.tags }}-py${{ matrix.python_version }}-${{ matrix.cuda }}-${{ matrix.pytorch }}${{ matrix.axolotl_extras != '' && '-' || '' }}${{ matrix.axolotl_extras }}
@@ -82,11 +77,6 @@ jobs:
             python_version: "3.9"
             pytorch: 2.0.1
             axolotl_extras: gptq
-          - cuda: 117
-            cuda_version: 11.7.1
-            python_version: "3.9"
-            pytorch: 1.13.1
-            axolotl_extras:
     runs-on: self-hosted
     steps:
       - name: Checkout

             python_version: "3.9"
             pytorch: 2.0.1
             axolotl_extras: gptq
     runs-on: self-hosted
     steps:
       - name: Checkout
         with:
           context: .
           build-args: |
+            BASE_TAG=${{ github.ref_name }}-base-py${{ matrix.python_version }}-${{ matrix.cuda }}-${{ matrix.pytorch }}
           file: ./docker/Dockerfile
           push: ${{ github.event_name != 'pull_request' }}
           tags: ${{ steps.metadata.outputs.tags }}-py${{ matrix.python_version }}-${{ matrix.cuda }}-${{ matrix.pytorch }}${{ matrix.axolotl_extras != '' && '-' || '' }}${{ matrix.axolotl_extras }}
             python_version: "3.9"
             pytorch: 2.0.1
             axolotl_extras: gptq
     runs-on: self-hosted
     steps:
       - name: Checkout

docker/Dockerfile-base CHANGED Viewed

@@ -8,7 +8,7 @@ FROM nvidia/cuda:$CUDA_VERSION-cudnn$CUDNN_VERSION-devel-ubuntu$UBUNTU_VERSION a
 ENV PATH="/root/miniconda3/bin:${PATH}"
 ARG PYTHON_VERSION="3.9"
-ARG PYTORCH="2.0.0"
 ARG CUDA="118"
 ENV PYTHON_VERSION=$PYTHON_VERSION
@@ -29,18 +29,18 @@ ENV PATH="/root/miniconda3/envs/py${PYTHON_VERSION}/bin:${PATH}"
 WORKDIR /workspace
 RUN python3 -m pip install --upgrade pip && pip3 install packaging && \
-    python3 -m pip install --no-cache-dir -U torch==${PYTORCH} torchvision torchaudio --extra-index-url https://download.pytorch.org/whl/cu$CUDA
 FROM base-builder AS flash-attn-builder
 WORKDIR /workspace
-ARG TORCH_CUDA_ARCH_LIST="7.0 7.5 8.0 8.6+PTX"
 RUN git clone https://github.com/Dao-AILab/flash-attention.git && \
     cd flash-attention && \
-    git checkout 9ee0ff1  && \
     python3 setup.py bdist_wheel && \
     cd csrc/fused_dense_lib && \
     python3 setup.py bdist_wheel && \
@@ -53,7 +53,7 @@ RUN git clone https://github.com/Dao-AILab/flash-attention.git && \
 FROM base-builder AS deepspeed-builder
-ARG TORCH_CUDA_ARCH_LIST="7.0 7.5 8.0 8.6+PTX"
 WORKDIR /workspace
@@ -74,6 +74,9 @@ RUN git clone https://github.com/TimDettmers/bitsandbytes.git && \
 FROM base-builder
 # recompile apex
 RUN python3 -m pip uninstall -y apex
 RUN git clone https://github.com/NVIDIA/apex

 ENV PATH="/root/miniconda3/bin:${PATH}"
 ARG PYTHON_VERSION="3.9"
+ARG PYTORCH_VERSION="2.0.1"
 ARG CUDA="118"
 ENV PYTHON_VERSION=$PYTHON_VERSION
 WORKDIR /workspace
 RUN python3 -m pip install --upgrade pip && pip3 install packaging && \
+    python3 -m pip install --no-cache-dir -U torch==${PYTORCH_VERSION}+cu${CUDA} --extra-index-url https://download.pytorch.org/whl/cu$CUDA
 FROM base-builder AS flash-attn-builder
 WORKDIR /workspace
+ARG TORCH_CUDA_ARCH_LIST="7.0 7.5 8.0 8.6 9.0+PTX"
 RUN git clone https://github.com/Dao-AILab/flash-attention.git && \
     cd flash-attention && \
+    git checkout v2.0.1  && \
     python3 setup.py bdist_wheel && \
     cd csrc/fused_dense_lib && \
     python3 setup.py bdist_wheel && \
 FROM base-builder AS deepspeed-builder
+ARG TORCH_CUDA_ARCH_LIST="7.0 7.5 8.0 8.6 9.0+PTX"
 WORKDIR /workspace
 FROM base-builder
+ARG TORCH_CUDA_ARCH_LIST="7.0 7.5 8.0 8.6 9.0+PTX"
+ENV TORCH_CUDA_ARCH_LIST=$TORCH_CUDA_ARCH_LIST
 # recompile apex
 RUN python3 -m pip uninstall -y apex
 RUN git clone https://github.com/NVIDIA/apex