improve handling and error if fa3 requested but not installeD

handle args to drop dropout
move fa3 tests to multigpu since we only run those on hopper
2025-05-19 10:11:14 -07:00 · 2025-05-18 15:17:40 -07:00 · 2025-05-18 15:17:39 -07:00 · 2025-05-18 15:17:39 -07:00 · 2025-05-18 15:17:39 -07:00 · 2025-05-18 15:17:39 -07:00
9 changed files with 108 additions and 18 deletions
--- a/.github/workflows/base.yml
+++ b/.github/workflows/base.yml
@@ -47,11 +47,18 @@ jobs:
            pytorch: 2.7.0
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
          - cuda: "128"
-            cuda_version: 12.6.3
+            cuda_version: 12.8.1
            cudnn_version: ""
            python_version: "3.11"
            pytorch: 2.7.0
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
+          - cuda: "126"
+            cuda_version: 12.6.3
+            cudnn_version: ""
+            python_version: "3.11"
+            pytorch: 2.6.0
+            suffix: "-hopper"
+            torch_cuda_arch_list: "9.0+PTX"
          - cuda: "128"
            cuda_version: 12.8.1
            cudnn_version: ""
@@ -87,7 +94,7 @@ jobs:
          context: .
          file: ${{ matrix.pytorch == 'nightly' && './docker/Dockerfile-base-nightly' || matrix.pytorch == 'next' && './docker/Dockerfile-base-next' || './docker/Dockerfile-base' }}
          push: ${{ github.event_name != 'pull_request' }}
-          tags: ${{ steps.metadata.outputs.tags }}-base-py${{ matrix.python_version }}-cu${{ matrix.cuda }}-${{ matrix.pytorch }}${{ matrix.axolotl_extras != '' && '-' || '' }}${{ matrix.axolotl_extras }}
+          tags: ${{ steps.metadata.outputs.tags }}-base-py${{ matrix.python_version }}-cu${{ matrix.cuda }}-${{ matrix.pytorch }}${{ matrix.axolotl_extras != '' && '-' || '' }}${{ matrix.axolotl_extras }}${{ matrix.suffix || '' }}
          labels: ${{ steps.metadata.outputs.labels }}
          build-args: |
            CUDA_VERSION=${{ matrix.cuda_version }}
--- a/.github/workflows/multi-gpu-e2e.yml
+++ b/.github/workflows/multi-gpu-e2e.yml
@@ -32,21 +32,25 @@ jobs:
            pytorch: 2.6.0
            axolotl_extras: vllm
            num_gpus: 2
-            nightly_build: "true"
+          - cuda: 126
+            cuda_version: 12.6.3
+            python_version: "3.11"
+            pytorch: 2.6.0
+            axolotl_extras:
+            suffix: "-hopper"
+            num_gpus: 2
          - cuda: 124
            cuda_version: 12.4.1
            python_version: "3.11"
            pytorch: 2.5.1
            axolotl_extras:
            num_gpus: 2
-            nightly_build: "true"
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
            pytorch: 2.7.0
            axolotl_extras:
            num_gpus: 2
-            nightly_build: "true"
    runs-on: [self-hosted, modal]
    timeout-minutes: 120
    steps:
@@ -68,7 +72,6 @@ jobs:
          echo "AXOLOTL_EXTRAS=${{ matrix.axolotl_extras}}" >> $GITHUB_ENV
          echo "CUDA=${{ matrix.cuda }}" >> $GITHUB_ENV
          echo "N_GPUS=${{ matrix.num_gpus }}" >> $GITHUB_ENV
-          echo "NIGHTLY_BUILD=${{ matrix.nightly_build }}" >> $GITHUB_ENV
          echo "CODECOV_TOKEN=${{ secrets.CODECOV_TOKEN }}" >> $GITHUB_ENV
      - name: Run tests job on Modal
        run: |
--- a/cicd/Dockerfile.jinja
+++ b/cicd/Dockerfile.jinja
@@ -32,6 +32,11 @@ RUN if [ "$NIGHTLY_BUILD" = "true" ] ; then \
    fi

 RUN pip install packaging==23.2 setuptools==75.8.0
+RUN if [ "$PYTORCH_VERSION" = "2.6.0" ] && [ "$CUDA" = "126" ] ; then \
+        curl -L -O https://d1dttdx32dkk5p.cloudfront.net/fa3/cu${CUDA}/torch-${PYTORCH_VERSION}/flash_attn_3-3.0.0b1-cp311-cp311-linux_x86_64.whl; \
+        pip3 install --no-cache-dir flash_attn_3-3.0.0b1-cp311-cp311-linux_x86_64.whl; \
+        rm flash_attn_3-3.0.0b1-cp311-cp311-linux_x86_64.whl; \
+    fi
 RUN if [ "$AXOLOTL_EXTRAS" != "" ] ; then \
        pip install --no-build-isolation -e .[deepspeed,flash-attn,ring-flash-attn,optimizers,ray,$AXOLOTL_EXTRAS] $AXOLOTL_ARGS; \
    else \
--- a/docker/Dockerfile-base
+++ b/docker/Dockerfile-base
@@ -1,5 +1,5 @@
-ARG CUDA_VERSION="11.8.0"
-ARG CUDNN_VERSION="8"
+ARG CUDA_VERSION="12.4.1"
+ARG CUDNN_VERSION=""
 ARG UBUNTU_VERSION="22.04"
 ARG MAX_JOBS=4

@@ -7,16 +7,16 @@ FROM nvidia/cuda:$CUDA_VERSION-cudnn$CUDNN_VERSION-devel-ubuntu$UBUNTU_VERSION A

 ENV PATH="/root/miniconda3/bin:${PATH}"

-ARG PYTHON_VERSION="3.10"
-ARG PYTORCH_VERSION="2.1.2"
-ARG CUDA="118"
+ARG PYTHON_VERSION="3.11"
+ARG PYTORCH_VERSION="2.5.1"
+ARG CUDA="124"
 ARG TORCH_CUDA_ARCH_LIST="7.0 7.5 8.0 8.6 9.0+PTX"

 ENV PYTHON_VERSION=$PYTHON_VERSION
 ENV TORCH_CUDA_ARCH_LIST=$TORCH_CUDA_ARCH_LIST

 RUN apt-get update \
-    && apt-get install -y wget git build-essential ninja-build git-lfs libaio-dev pkg-config && rm -rf /var/lib/apt/lists/* \
+    && apt-get install -y wget git build-essential ninja-build git-lfs libaio-dev pkg-config curl && rm -rf /var/lib/apt/lists/* \
    && wget \
    https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh \
    && mkdir /root/.conda \
@@ -38,6 +38,10 @@ RUN git lfs install --skip-repo && \
    # The base image ships with `pydantic==1.8.2` which is not working
    pip3 install -U --no-cache-dir pydantic==1.10.10

-RUN if [ "$PYTORCH_VERSION" = "2.7.0" ] ; then \
+RUN if [ "$TORCH_CUDA_ARCH_LIST" = "9.0+PTX" ] ; then \
+        curl -L -O https://d1dttdx32dkk5p.cloudfront.net/fa3/cu${CUDA}/torch-${PYTORCH_VERSION}/flash_attn_3-3.0.0b1-cp311-cp311-linux_x86_64.whl; \
+        pip3 install --no-cache-dir flash_attn_3-3.0.0b1-cp311-cp311-linux_x86_64.whl; \
+        rm flash_attn_3-3.0.0b1-cp311-cp311-linux_x86_64.whl; \
+    elif [ "$PYTORCH_VERSION" = "2.7.0" ] ; then \
        pip3 install flash-attn==2.7.4.post1; \
    fi
--- a/src/axolotl/utils/models.py
+++ b/src/axolotl/utils/models.py
@@ -629,6 +629,49 @@ class ModelLoader:
            )

        if self.cfg.flash_attention:
+            use_fa3 = False
+            if self.cfg.use_flash_attention_3 is True:
+                use_fa3 = True
+            elif self.cfg.use_flash_attention_3 == "auto":
+                if torch.cuda.get_device_capability() >= (9, 0):
+                    # FA3 is only available on Hopper GPUs and newer
+                    use_fa3 = True
+                if not importlib.util.find_spec("flash_attn_interface"):
+                    use_fa3 = False
+            if use_fa3 and not importlib.util.find_spec("flash_attn_interface"):
+                # this can happen when use_flash_attention_3 is explicity set to True
+                # and flash_attn_interface is not installed
+                raise ModuleNotFoundError(
+                    "Please install the flash_attn_interface library to use Flash Attention 3.x"
+                )
+            if use_fa3 and importlib.util.find_spec("flash_attn_interface") is not None:
+                from flash_attn_interface import flash_attn_func as flash_attn_func_v3
+                from flash_attn_interface import (
+                    flash_attn_varlen_func as flash_attn_varlen_func_v3,
+                )
+
+                def flash_attn_func_v3_wrapper(*args, **kwargs):
+                    kwargs.pop("dropout_p", None)
+                    if "softmax_scale" in kwargs and len(args) >= 4:
+                        # if softmax_scale is provided, then the 3rd position is dropout_p that we need to drop
+                        args = (*args[:3],) + args[4:]
+                    return flash_attn_func_v3(*args, **kwargs)[0]
+
+                def flash_attn_varlen_func_v3_wrapper(*args, **kwargs):
+                    kwargs.pop("dropout_p", None)
+                    if "softmax_scale" in kwargs and len(args) >= 4:
+                        # if softmax_scale is provided, then the 3rd position is dropout_p that we need to drop
+                        args = (*args[:3],) + args[4:]
+                    return flash_attn_varlen_func_v3(*args, **kwargs)[0]
+
+                transformers.modeling_flash_attention_utils.flash_attn_func = (
+                    flash_attn_func_v3_wrapper
+                )
+                transformers.modeling_flash_attention_utils.flash_attn_varlen_func = (
+                    flash_attn_varlen_func_v3_wrapper
+                )
+                LOG.info("Switched to Flash Attention v3")
+
            self.patch_attention()

        if self.cfg.sample_packing and self.cfg.s2_attention:
@@ -699,6 +742,7 @@ class ModelLoader:

                patch_mllama()

+            # TODO deprecate soon
            if self.model_config.model_type == "btlm":
                from axolotl.monkeypatch.btlm_attn_hijack_flash import (
                    replace_btlm_attn_with_flash_attn,
@@ -706,6 +750,7 @@ class ModelLoader:

                replace_btlm_attn_with_flash_attn(self.cfg.base_model)

+            # TODO deprecate soon
            if (
                self.model_config.model_type == "stablelm_epoch"
                and self.cfg.sample_packing
--- a/src/axolotl/utils/schemas/config.py
+++ b/src/axolotl/utils/schemas/config.py
@@ -233,6 +233,7 @@ class AxolotlInputConfig(
    flash_attn_fuse_qkv: bool | None = None
    flash_attn_fuse_mlp: bool | None = None
    flash_optimum: bool | None = None
+    use_flash_attention_3: Literal["auto"] | bool | None = None

    eager_attention: bool | None = None

--- a/tests/conftest.py
+++ b/tests/conftest.py
@@ -421,6 +421,7 @@ def temp_dir():

@pytest.fixture(scope="function", autouse=True)
 def cleanup_monkeypatches():
+    import transformers.modeling_flash_attention_utils
    from transformers import Trainer
    from transformers.models.llama.modeling_llama import (  # LlamaFlashAttention2,
        LlamaAttention,
@@ -434,6 +435,19 @@ def cleanup_monkeypatches():
        Trainer._inner_training_loop  # pylint: disable=protected-access
    )
    original_trainer_training_step = Trainer.training_step
+    original_fa_func = None
+    original_fa_varlen_func = None
+    if (
+        importlib.util.find_spec("flash_attn")
+        and hasattr(transformers.modeling_flash_attention_utils, "flash_attn_func")
+        and hasattr(
+            transformers.modeling_flash_attention_utils, "flash_attn_varlen_func"
+        )
+    ):
+        original_fa_func = transformers.modeling_flash_attention_utils.flash_attn_func
+        original_fa_varlen_func = (
+            transformers.modeling_flash_attention_utils.flash_attn_varlen_func
+        )
    # monkey patches can happen inside the tests
    yield
    # Reset LlamaFlashAttention2 forward
@@ -444,6 +458,11 @@ def cleanup_monkeypatches():
        original_trainer_inner_training_loop
    )
    Trainer.training_step = original_trainer_training_step
+    if original_fa_func:
+        transformers.modeling_flash_attention_utils.flash_attn_func = original_fa_func
+        transformers.modeling_flash_attention_utils.flash_attn_varlen_func = (
+            original_fa_varlen_func
+        )

    # Reset other known monkeypatches
    modules_to_reset: list[tuple[str, list[str]]] = [
@@ -458,6 +477,7 @@ def cleanup_monkeypatches():
        ("transformers.trainer",),
        ("transformers", ["Trainer"]),
        ("transformers.loss.loss_utils",),
+        ("transformers.modeling_flash_attention_utils",),
    ]
    for module_name_tuple in modules_to_reset:
        module_name = module_name_tuple[0]
--- a/tests/e2e/multigpu/test_llama.py
+++ b/tests/e2e/multigpu/test_llama.py
@@ -101,7 +101,13 @@ class TestMultiGPULlama:
        "gradient_accumulation_steps",
        [1, 2],
    )
-    def test_lora_ddp_packed(self, temp_dir, gradient_accumulation_steps):
+    @pytest.mark.parametrize(
+        "use_flash_attention_3",
+        [False, "auto"],
+    )
+    def test_lora_ddp_packed(
+        self, temp_dir, gradient_accumulation_steps, use_flash_attention_3
+    ):
        # pylint: disable=duplicate-code
        cfg = DictDefault(
            {
@@ -138,6 +144,7 @@ class TestMultiGPULlama:
                "flash_attention": True,
                "use_tensorboard": True,
                "bf16": True,
+                "use_flash_attention_3": use_flash_attention_3,
            }
        )

--- a/tests/e2e/test_packing_loss.py
+++ b/tests/e2e/test_packing_loss.py
@@ -4,7 +4,6 @@ E2E tests for packed training

 import logging
 import os
-import unittest

 from transformers.utils import is_torch_bf16_gpu_available

@@ -14,18 +13,17 @@ from axolotl.train import train
 from axolotl.utils.config import normalize_config, validate_config
 from axolotl.utils.dict import DictDefault

-from .utils import check_tensorboard, with_temp_dir
+from .utils import check_tensorboard

 LOG = logging.getLogger("axolotl.tests.e2e")
 os.environ["WANDB_DISABLED"] = "true"


-class TestPackedLlama(unittest.TestCase):
+class TestPackedLlama:
    """
    Test case for Packed training of llama models
    """

-    @with_temp_dir
    def test_loss_packed(self, temp_dir):
        # pylint: disable=duplicate-code
        cfg = DictDefault(
Author	SHA1	Message	Date
Wing Lian	9bdf4b1c23	improve handling and error if fa3 requested but not installeD	2025-05-19 10:11:14 -07:00
Wing Lian	d6f64a3684	handle args to drop dropout	2025-05-18 15:17:40 -07:00
Wing Lian	0735454782	move fa3 tests to multigpu since we only run those on hopper	2025-05-18 15:17:39 -07:00
Wing Lian	bb6464c4c6	use get_device_capability since CI setting in cfg is unreliable	2025-05-18 15:17:39 -07:00
Wing Lian	323a9cb153	handle return sig change for fa3	2025-05-18 15:17:39 -07:00
Wing Lian	b22150751f	check for fa first	2025-05-18 15:17:39 -07:00
Wing Lian	8c4bc59bfc	fa3 doesn't support dropout_p, fix unpatching	2025-05-18 15:17:39 -07:00
Wing Lian	a064f1c9b4	ci for fa3	2025-05-18 15:17:39 -07:00
Wing Lian	fb5ef6d445	use updated package name for fa3	2025-05-18 15:17:38 -07:00
Wing Lian	34b68ddaae	curl with apt instead of pip	2025-05-18 15:17:38 -07:00
Wing Lian	9a3d0c919b	make sure curl is installed	2025-05-18 15:17:38 -07:00
Wing Lian	bd34d0b861	install for hopper from pre-built wheel	2025-05-18 15:17:38 -07:00
Wing Lian	37220ab90a	install pybind11 for fa3 build	2025-05-18 15:17:38 -07:00
Wing Lian	e1b74d710b	update docker args to minimums used and use MAX_JOBS already set as arg	2025-05-18 15:17:38 -07:00
Wing Lian	79daf5b934	reduce max jobs for build of fa3	2025-05-18 15:17:38 -07:00
Wing Lian	ddd7c55576	build hopper w fa3 on torch 2.6	2025-05-18 15:17:37 -07:00
Wing Lian	65c6c98a76	whitespace fix in dockerfile	2025-05-18 15:17:37 -07:00
Wing Lian	4ef2e8293f	fix the bash in docker base	2025-05-18 15:17:37 -07:00
Wing Lian	c126d5cd04	fix suffix for tag	2025-05-18 15:17:37 -07:00
Wing Lian	9b0be4f15c	fix 12.8 image and add flash-attn v3 hopper base image	2025-05-18 15:17:37 -07:00