update fsdp2 patch

workaround for fsdp2 optimizer save failures
use latest transformers on main with fix
2025-07-23 16:53:03 +01:00 · 2025-07-23 09:38:57 -04:00 · 2025-07-23 09:22:36 -04:00 · 2025-07-23 08:43:34 -04:00 · 2025-07-22 21:21:45 -04:00 · 2025-07-22 21:20:57 -04:00
538 changed files with 16575 additions and 23051 deletions
--- a/.bandit
+++ b/.bandit
@@ -1,3 +1,3 @@
 [bandit]
 exclude = tests
-skips = B101,B615,B102,B110
+skips = B101,B615
--- a/.coderabbit.yaml
+++ b/.coderabbit.yaml
@@ -12,6 +12,5 @@ reviews:
  auto_review:
    enabled: true
    drafts: false
    auto_incremental_review: false
 chat:
  auto_reply: true
--- a/.flake8
+++ b/.flake8
@@ -0,0 +1,5 @@
 [flake8]
 max-line-length = 88
 select = C,E,F,W,B,B950
 extend-ignore = E203, E501, W503
--- a/.github/CONTRIBUTING.md
+++ b/.github/CONTRIBUTING.md
@@ -57,13 +57,6 @@ We welcome ideas for improvements and new features. To suggest an enhancement, o
 5. Push your branch to your fork on GitHub.
 6. Open a new pull request against the `main` branch of the axolotl repository. Include a clear and concise description of your changes, referencing any related issues.
 #### Skipping CI Checks
 You can skip certain CI checks by including specific keywords in your commit messages:
 - `[skip ci]` or `skip ci` - Skips all CI checks for that commit
 - `[skip-e2e]` or `skip-e2e` - Skips only end-to-end tests while running other CI checks. You may also include this in the title of your PR to disable end-to-end tests for the entire PR.
 ## Style Guidelines
 ### Code Style
--- a/.github/workflows/base.yml
+++ b/.github/workflows/base.yml
@@ -17,7 +17,7 @@ on:
 jobs:
  build-base:
-    if: ${{ github.repository_owner == 'axolotl-ai-cloud' && (github.event_name != 'pull_request' || !github.event.pull_request.draft) }}
+    if: github.repository_owner == 'axolotl-ai-cloud'
    timeout-minutes: 480
    # this job needs to be run on self-hosted GPU runners...
    runs-on: ubuntu-latest-m
@@ -54,7 +54,7 @@ jobs:
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
            dockerfile: "Dockerfile-base"
          - cuda: "128"
-            cuda_version: 12.8.1
+            cuda_version: 12.6.3
            cudnn_version: ""
            python_version: "3.11"
            pytorch: 2.7.1
@@ -64,16 +64,9 @@ jobs:
            cuda_version: 12.8.1
            cudnn_version: ""
            python_version: "3.11"
-            pytorch: 2.8.0
+            pytorch: nightly
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
-            dockerfile: "Dockerfile-base"
+            dockerfile: "Dockerfile-base-nightly"
 #          - cuda: "128"
 #            cuda_version: 12.8.1
 #            cudnn_version: ""
 #            python_version: "3.11"
 #            pytorch: nightly
 #            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
 #            dockerfile: "Dockerfile-base-nightly"
 #          # "next" is for release candidates of pytorch
 #          - cuda: "128"
 #            cuda_version: 12.8.1
@@ -115,7 +108,7 @@ jobs:
            PYTORCH_VERSION=${{ matrix.pytorch }}
            TORCH_CUDA_ARCH_LIST=${{ matrix.torch_cuda_arch_list }}
  build-base-uv:
-    if: ${{ github.repository_owner == 'axolotl-ai-cloud' && (github.event_name != 'pull_request' || !github.event.pull_request.draft) }}
+    if: github.repository_owner == 'axolotl-ai-cloud'
    timeout-minutes: 480
    runs-on: ubuntu-latest-m
    strategy:
@@ -129,13 +122,6 @@ jobs:
            pytorch: 2.6.0
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
            dockerfile: "Dockerfile-uv-base"
          - cuda: "126"
            cuda_version: 12.6.3
            cudnn_version: ""
            python_version: "3.11"
            pytorch: 2.7.1
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
            dockerfile: "Dockerfile-uv-base"
          - cuda: "128"
            cuda_version: 12.8.1
            cudnn_version: ""
@@ -143,13 +129,6 @@ jobs:
            pytorch: 2.7.1
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
            dockerfile: "Dockerfile-uv-base"
          - cuda: "128"
            cuda_version: 12.8.1
            cudnn_version: ""
            python_version: "3.11"
            pytorch: 2.8.0
            torch_cuda_arch_list: "7.0 7.5 8.0 8.6 8.7 8.9 9.0+PTX"
            dockerfile: "Dockerfile-uv-base"
    steps:
      - name: Checkout
        uses: actions/checkout@v4
--- a/.github/workflows/lint.yml
+++ b/.github/workflows/lint.yml
@@ -3,7 +3,6 @@ on:
  # check on PRs, and manual triggers
  merge_group:
  pull_request:
      types: [opened, synchronize, reopened, ready_for_review]
      paths:
       - '**.py'
       - 'requirements.txt'
@@ -17,7 +16,6 @@ jobs:
  pre-commit:
    name: pre-commit
    runs-on: ubuntu-latest
    if: ${{ !github.event.pull_request.draft }}
    steps:
      - uses: actions/checkout@v4
      - uses: actions/setup-python@v5
--- a/.github/workflows/main.yml
+++ b/.github/workflows/main.yml
@@ -24,22 +24,16 @@ jobs:
            cuda_version: 12.6.3
            python_version: "3.11"
            pytorch: 2.7.0
-            axolotl_extras:
+            axolotl_extras: vllm
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras: vllm
            is_latest: true
          - cuda: 128
            cuda_version: 12.8.1
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras:
          - cuda: 128
            cuda_version: 12.8.1
            python_version: "3.11"
-            pytorch: 2.8.0
+            pytorch: 2.7.1
            axolotl_extras:
    runs-on: axolotl-gpu-runner
    steps:
@@ -103,23 +97,12 @@ jobs:
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras:
            is_latest:
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras: vllm
            is_latest: true
          - cuda: 128
            cuda_version: 12.8.1
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras:
          - cuda: 128
            cuda_version: 12.8.1
            python_version: "3.11"
            pytorch: 2.8.0
            axolotl_extras:
    runs-on: axolotl-gpu-runner
    steps:
      - name: Checkout
@@ -167,24 +150,6 @@ jobs:
            python_version: "3.11"
            pytorch: 2.6.0
            axolotl_extras:
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras:
            is_latest:
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras: vllm
            is_latest: true
          - cuda: 128
            cuda_version: 12.8.1
            python_version: "3.11"
            pytorch: 2.8.0
            axolotl_extras:
            is_latest:
    runs-on: axolotl-gpu-runner
    steps:
      - name: Checkout
--- a/.github/workflows/multi-gpu-e2e.yml
+++ b/.github/workflows/multi-gpu-e2e.yml
@@ -21,7 +21,7 @@ concurrency:
 jobs:
  test-axolotl-multigpu:
-    if: ${{ ! contains(github.event.commits[0].message, '[skip e2e]') && github.repository_owner == 'axolotl-ai-cloud' && (github.event_name != 'pull_request' || !github.event.pull_request.draft) }}
+    if: ${{ ! contains(github.event.commits[0].message, '[skip e2e]') && github.repository_owner == 'axolotl-ai-cloud' }}
    strategy:
      fail-fast: false
      matrix:
@@ -36,15 +36,15 @@ jobs:
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
-            pytorch: 2.7.1
+            pytorch: 2.7.0
            axolotl_extras: vllm
            num_gpus: 2
            nightly_build: "true"
-          - cuda: 128
+          - cuda: 126
-            cuda_version: 12.8.1
+            cuda_version: 12.6.3
            python_version: "3.11"
-            pytorch: 2.8.0
+            pytorch: 2.7.1
-            axolotl_extras: fbgemm-gpu
+            axolotl_extras:
            num_gpus: 2
            nightly_build: "true"
    runs-on: [self-hosted, modal]
--- a/.github/workflows/preview-docs.yml
+++ b/.github/workflows/preview-docs.yml
@@ -2,7 +2,7 @@ name: Preview
 on:
  workflow_dispatch:
  pull_request:
-    types: [opened, synchronize, reopened, ready_for_review]
+    types: [opened, synchronize, reopened]
    # Run the workflow only when one of these files changes
    paths:
@@ -25,7 +25,6 @@ permissions:
 jobs:
  preview:
    runs-on: ubuntu-latest
    if: ${{ !github.event.pull_request.draft }}
    steps:
      - name: Check out repository
        uses: actions/checkout@v4
@@ -53,7 +52,6 @@ jobs:
      - name: Netlify Publish
        uses: nwtgck/actions-netlify@v3.0
        if: ${{ github.event.pull_request.head.repo.full_name == github.repository }}
        id: netlify
        with:
          publish-dir: './_site'
--- a/.github/workflows/tests.yml
+++ b/.github/workflows/tests.yml
@@ -13,7 +13,6 @@ on:
      - 'cicd/cicd.sh'
      - 'cicd/Dockerfile.jinja'
  pull_request:
      types: [opened, synchronize, reopened, ready_for_review]
      paths:
       - '**.py'
       - 'requirements.txt'
@@ -35,7 +34,6 @@ jobs:
  pre-commit:
    name: pre-commit
    runs-on: ubuntu-latest
    if: ${{ !github.event.pull_request.draft }}
    steps:
      - uses: actions/checkout@v4
      - uses: actions/setup-python@v5
@@ -49,13 +47,12 @@ jobs:
  pytest:
    name: PyTest
    runs-on: ubuntu-latest
    if: ${{ !github.event.pull_request.draft }}
 #    needs: [preload-cache]
    strategy:
      fail-fast: false
      matrix:
        python_version: ["3.11"]
-        pytorch_version: ["2.6.0", "2.7.1", "2.8.0"]
+        pytorch_version: ["2.6.0", "2.7.0", "2.7.1"]
    timeout-minutes: 20
    steps:
@@ -105,8 +102,7 @@ jobs:
      - name: Run tests
        run: |
-          pytest -v --durations=10 -n8 --dist loadfile --ignore=tests/e2e/ --ignore=tests/patched/ --ignore=tests/cli/ --ignore=tests/monkeypatch/ tests/ --cov=axolotl --cov-report=xml
+          pytest -v --durations=10 -n8 --dist loadfile --ignore=tests/e2e/ --ignore=tests/patched/ --ignore=tests/cli/ tests/ --cov=axolotl --cov-report=xml
          pytest -v --durations=10 tests/monkeypatch/ --cov=axolotl --cov-append --cov-report=xml
          pytest -v --durations=10 tests/patched/ --cov=axolotl --cov-append --cov-report=xml
          pytest -v --durations=10 tests/cli/ --cov=axolotl --cov-append --cov-report=xml
@@ -125,12 +121,11 @@ jobs:
  pytest-sdist:
    name: PyTest from Source Dist
    runs-on: ubuntu-latest
    if: ${{ !github.event.pull_request.draft }}
    strategy:
      fail-fast: false
      matrix:
        python_version: ["3.11"]
-        pytorch_version: ["2.6.0", "2.7.1", "2.8.0"]
+        pytorch_version: ["2.6.0", "2.7.0", "2.7.1"]
    timeout-minutes: 20
    steps:
@@ -180,52 +175,21 @@ jobs:
      - name: Run tests
        run: |
-          pytest -v --durations=10 -n8 --dist loadfile --ignore=tests/e2e/ --ignore=tests/patched/ --ignore=tests/cli/ --ignore=tests/monkeypatch/ tests/ --cov=axolotl --cov-report=xml
+          pytest -v --durations=10 -n8 --dist loadfile --ignore=tests/e2e/ --ignore=tests/patched/ --ignore=tests/cli/ tests/
-          pytest -v --durations=10 tests/monkeypatch/ --cov=axolotl --cov-append --cov-report=xml
+          pytest -v --durations=10 tests/patched/
          pytest -v --durations=10 tests/cli/
      - name: cleanup pip cache
        run: |
          find "$(pip cache dir)/http-v2" -type f -mtime +14 -exec rm {} \;
  gate-skip-e2e:
    needs: [pre-commit, pytest, pytest-sdist]
    runs-on: ubuntu-latest
    outputs:
      skip: ${{ steps.compute.outputs.skip }}
    steps:
      - uses: actions/github-script@v7
        id: compute
        with:
          script: |
            const token = /\[skip-e2e\]/i;
            let msg = '';
            if (context.eventName === 'push') {
              msg = context.payload.head_commit?.message || '';
            } else if (context.eventName === 'pull_request') {
              const { owner, repo } = context.repo;
              const prNumber = context.payload.pull_request.number;
              const commits = await github.paginate(
                github.rest.pulls.listCommits,
                { owner, repo, pull_number: prNumber, per_page: 100 }
              );
              msg = commits.at(-1)?.commit?.message || '';
            }
            const title = context.payload.pull_request?.title || '';
            const body  = context.payload.pull_request?.body  || '';
            const skip = token.test(msg) || token.test(title) || token.test(body);
            core.setOutput('skip', String(skip));
  docker-e2e-tests-1st:
    # Run this job first as a gate for running the remainder of the test matrix
-    if: >
+    if: ${{ ! contains(github.event.commits[0].message, '[skip e2e]') && github.repository_owner == 'axolotl-ai-cloud' }}
      github.repository_owner == 'axolotl-ai-cloud' &&
      (github.event_name != 'pull_request' || !github.event.pull_request.draft) &&
      needs.gate-skip-e2e.outputs.skip != 'true'
    # this job needs to be run on self-hosted GPU runners...
    runs-on: [self-hosted, modal]
    timeout-minutes: 120
-    needs: [pre-commit, pytest, pytest-sdist, gate-skip-e2e]
+    needs: [pre-commit, pytest, pytest-sdist]
    strategy:
      fail-fast: false
@@ -240,7 +204,7 @@ jobs:
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
-            pytorch: 2.7.1
+            pytorch: 2.6.0
            num_gpus: 1
            axolotl_extras:
            dockerfile: "Dockerfile-uv.jinja"
@@ -271,16 +235,13 @@ jobs:
          modal run cicd.e2e_tests
  docker-e2e-tests:
-    if: >
+    if: github.repository_owner == 'axolotl-ai-cloud'
      github.repository_owner == 'axolotl-ai-cloud' &&
      (github.event_name != 'pull_request' || !github.event.pull_request.draft) &&
      needs.gate-skip-e2e.outputs.skip != 'true'
    # this job needs to be run on self-hosted GPU runners...
    runs-on: [self-hosted, modal]
    timeout-minutes: 120
    # Only run the remainder of the matrix if the first e2e check passed;
    # this is to save on wasted compute costs for known failures that get caught in the first run
-    needs: [pre-commit, pytest, gate-skip-e2e, docker-e2e-tests-1st]
+    needs: [pre-commit, pytest, docker-e2e-tests-1st]
    strategy:
      fail-fast: false
@@ -298,13 +259,6 @@ jobs:
            pytorch: 2.7.1
            num_gpus: 1
            axolotl_extras:
          - cuda: 128
            cuda_version: 12.8.1
            python_version: "3.11"
            pytorch: 2.8.0
            num_gpus: 1
            gpu_type: "B200"
            axolotl_extras: fbgemm-gpu
    steps:
      - name: Checkout
        uses: actions/checkout@v4
@@ -325,7 +279,6 @@ jobs:
          echo "CUDA=${{ matrix.cuda }}" >> $GITHUB_ENV
          echo "MODAL_IMAGE_BUILDER_VERSION=2024.10" >> $GITHUB_ENV
          echo "N_GPUS=${{ matrix.num_gpus }}" >> $GITHUB_ENV
          echo "GPU_TYPE=${{ matrix.gpu_type || 'L40S'}}" >> $GITHUB_ENV
          echo "CODECOV_TOKEN=${{ secrets.CODECOV_TOKEN }}" >> $GITHUB_ENV
          echo "E2E_DOCKERFILE=${{ matrix.dockerfile || 'Dockerfile.jinja'}}" >> $GITHUB_ENV
      - name: Run tests job on Modal
@@ -336,16 +289,15 @@ jobs:
    runs-on: [self-hosted, modal]
    timeout-minutes: 90
    needs: [docker-e2e-tests]
    if: ${{ !github.event.pull_request.draft }}
    strategy:
      fail-fast: false
      matrix:
        include:
-          - cuda: 126
+          - cuda: 124
-            cuda_version: 12.6.3
+            cuda_version: 12.4.1
            python_version: "3.11"
-            pytorch: 2.7.1
+            pytorch: 2.6.0
            num_gpus: 1
            axolotl_extras:
    steps:
--- a/.isort.cfg
+++ b/.isort.cfg
@@ -0,0 +1,4 @@
 [settings]
 profile=black
 known_third_party=wandb,comet_ml
 known_local_folder=src,tests
--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -3,21 +3,31 @@ default_language_version:
 repos:
 -   repo: https://github.com/pre-commit/pre-commit-hooks
-    rev: v6.0.0
+    rev: v5.0.0
    hooks:
    -   id: check-yaml
    -   id: end-of-file-fixer
    -   id: trailing-whitespace
    -   id: no-commit-to-branch
        args: ['--branch', 'main']
-   repo: https://github.com/astral-sh/ruff-pre-commit
+-   repo: https://github.com/psf/black
-    rev: v0.12.12
+    rev: 25.1.0
    hooks:
-    -   id: ruff
+    -   id: black
-        args: [--fix, --select, I]
+-   repo: https://github.com/pycqa/isort
-    -   id: ruff-format
+    rev: 6.0.1
    hooks:
      - id: isort
 -   repo: https://github.com/PyCQA/flake8
    rev: 7.3.0
    hooks:
    - id: flake8
 -   repo: https://github.com/pylint-dev/pylint
    rev: v3.3.7
    hooks:
    - id: pylint
 -   repo: https://github.com/pre-commit/mirrors-mypy
-    rev: v1.17.1
+    rev: v1.17.0
    hooks:
    - id: mypy
      additional_dependencies:
--- a/.pylintrc
+++ b/.pylintrc
@@ -0,0 +1,15 @@
 [MASTER]
 init-hook="from pylint.config import find_default_config_files; import sys; sys.path.append(next(find_default_config_files()).parent.as_posix())"
 [TYPECHECK]
 # List of members which are set dynamically and missed by Pylint inference
 # system, and so shouldn't trigger E1101 when accessed.
 generated-members=numpy.*, torch.*
 [pylint.messages_control]
 disable=missing-function-docstring, line-too-long, import-error,
    too-many-arguments, too-many-locals, too-many-statements, too-many-branches, too-few-public-methods,
    too-many-instance-attributes, fixme, import-outside-toplevel, logging-fstring-interpolation,
    too-many-positional-arguments, possibly-used-before-assignment
--- a/.runpod/README.md
+++ b/.runpod/README.md
@@ -119,15 +119,14 @@ datasets:
 ## Dataset Processing
-| Option                            | Default                    | Description                         |
+| Option                        | Default                    | Description                       |
-| --------------------------------- | -------------------------- | ----------------------------------- |
+| ----------------------------- | -------------------------- | --------------------------------- |
-| `dataset_prepared_path`           | `"data/last_run_prepared"` | Path for prepared dataset           |
+| `dataset_prepared_path`       | `"data/last_run_prepared"` | Path for prepared dataset         |
-| `push_dataset_to_hub`             | `""`                       | Push dataset to HF hub              |
+| `push_dataset_to_hub`         | `""`                       | Push dataset to HF hub            |
-| `dataset_processes`               | `4`                        | Number of preprocessing processes   |
+| `dataset_processes`           | `4`                        | Number of preprocessing processes |
-| `dataset_keep_in_memory`          | `false`                    | Keep dataset in memory              |
+| `dataset_keep_in_memory`      | `false`                    | Keep dataset in memory            |
-| `shuffle_merged_datasets`         | `true`                     | Shuffle merged datasets             |
+| `shuffle_merged_datasets`     | `true`                     | Shuffle merged datasets           |
-| `shuffle_before_merging_datasets` | `false`                    | Shuffle each dataset before merging |
+| `dataset_exact_deduplication` | `true`                     | Deduplicate datasets              |
 | `dataset_exact_deduplication`     | `true`                     | Deduplicate datasets                |
 ## LoRA Configuration
@@ -185,6 +184,7 @@ datasets:
 | `flash_attention`          | `false` | Use flash attention           |
 | `flash_attn_cross_entropy` | `false` | Flash attention cross entropy |
 | `flash_attn_rms_norm`      | `false` | Flash attention RMS norm      |
 | `flash_attn_fuse_qkv`      | `false` | Fuse QKV operations           |
 | `flash_attn_fuse_mlp`      | `false` | Fuse MLP operations           |
 | `sdp_attention`            | `false` | Use scaled dot product        |
 | `s2_attention`             | `false` | Use shifted sparse attention  |
--- a/.runpod/src/config/config.yaml
+++ b/.runpod/src/config/config.yaml
@@ -296,6 +296,7 @@
 # flash_attention:
 # flash_attn_cross_entropy:  # Whether to use flash-attention cross entropy implementation - advanced use only
 # flash_attn_rms_norm:  # Whether to use flash-attention rms norm implementation - advanced use only
 # flash_attn_fuse_qkv: # Whether to fuse QKV into a single operation
 # flash_attn_fuse_mlp: # Whether to fuse part of the MLP into a single operation
 # # Whether to use scaled-dot-product attention
 # # https://pytorch.org/docs/stable/generated/torch.nn.functional.scaled_dot_product_attention.html
@@ -540,6 +541,7 @@ xformers_attention: ${XFORMERS_ATTENTION}
 flash_attention: ${FLASH_ATTENTION}
 flash_attn_cross_entropy: ${FLASH_ATTN_CROSS_ENTROPY}
 flash_attn_rms_norm: ${FLASH_ATTN_RMS_NORM}
 flash_attn_fuse_qkv: ${FLASH_ATTN_FUSE_QKV}
 flash_attn_fuse_mlp: ${FLASH_ATTN_FUSE_MLP}
 sdp_attention: ${SDP_ATTENTION}
 s2_attention: ${S2_ATTENTION}
--- a/CITATION.cff
+++ b/CITATION.cff
@@ -1,10 +0,0 @@
 cff-version: 1.2.0
 type: software
 title: "Axolotl: Open Source LLM Post-Training"
 message: "If you use this software, please cite it as below."
 authors:
  - name: "Axolotl maintainers and contributors"
 repository-code: "https://github.com/axolotl-ai-cloud/axolotl"
 url: "https://axolotl.ai/"
 license: Apache-2.0
 date-released: "2023-05-30"
--- a/README.md
+++ b/README.md
@@ -5,9 +5,6 @@
        <img alt="Axolotl" src="https://raw.githubusercontent.com/axolotl-ai-cloud/axolotl/887513285d98132142bf5db2a74eb5e0928787f1/image/axolotl_logo_digital_black.svg" width="400" height="104" style="max-width: 100%;">
    </picture>
 </p>
  <p align="center">
      <strong>A Free and Open Source LLM Fine-tuning Framework</strong><br>
  </p>
 <p align="center">
    <img src="https://img.shields.io/github/license/axolotl-ai-cloud/axolotl.svg?color=blue" alt="GitHub License">
@@ -20,7 +17,6 @@
    <br/>
    <a href="https://discord.com/invite/HhrNrHJPRb"><img src="https://img.shields.io/badge/discord-7289da.svg?style=flat-square&logo=discord" alt="discord" style="height: 20px;"></a>
    <a href="https://twitter.com/axolotl_ai"><img src="https://img.shields.io/twitter/follow/axolotl_ai?style=social" alt="twitter" style="height: 20px;"></a>
    <a href="https://colab.research.google.com/github/axolotl-ai-cloud/axolotl/blob/main/examples/colab-notebooks/colab-axolotl-example.ipynb"><img src="https://colab.research.google.com/assets/colab-badge.svg" alt="google-colab" style="height: 20px;"></a>
    <br/>
    <img src="https://github.com/axolotl-ai-cloud/axolotl/actions/workflows/tests-nightly.yml/badge.svg" alt="tests-nightly">
    <img src="https://github.com/axolotl-ai-cloud/axolotl/actions/workflows/multi-gpu-e2e.yml/badge.svg" alt="multigpu-semi-weekly tests">
@@ -29,45 +25,31 @@
 ## 🎉 Latest Updates
 - 2025/07:
  - ND Parallelism support has been added into Axolotl. Compose Context Parallelism (CP), Tensor Parallelism (TP), and Fully Sharded Data Parallelism (FSDP) within a single node and across multiple nodes. Check out the [blog post](https://huggingface.co/blog/accelerate-nd-parallel) for more info.
  - Axolotl adds more models: [GPT-OSS](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/gpt-oss), [Gemma 3n](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/gemma3n), [Liquid Foundation Model 2 (LFM2)](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/lfm2), and [Arcee Foundation Models (AFM)](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/afm).
  - FP8 finetuning with fp8 gather op is now possible in Axolotl via `torchao`. Get started [here](https://docs.axolotl.ai/docs/mixed_precision.html#sec-fp8)!
  - [Voxtral](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/voxtral), [Magistral 1.1](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/magistral), and [Devstral](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/devstral) with mistral-common tokenizer support has been integrated in Axolotl!
  - TiledMLP support for single-GPU to multi-GPU training with DDP, DeepSpeed and FSDP support has been added to support Arctic Long Sequence Training. (ALST). See [examples](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/alst) for using ALST with Axolotl!
 - 2025/05: Quantization Aware Training (QAT) support has been added to Axolotl. Explore the [docs](https://docs.axolotl.ai/docs/qat.html) to learn more!
 - 2025/03: Axolotl has implemented Sequence Parallelism (SP) support. Read the [blog](https://huggingface.co/blog/axolotl-ai-co/long-context-with-sequence-parallelism-in-axolotl) and [docs](https://docs.axolotl.ai/docs/sequence_parallelism.html) to learn how to scale your context length when fine-tuning.
 <details>
 <summary>Expand older updates</summary>
 - 2025/06: Magistral with mistral-common tokenizer support has been added to Axolotl. See [examples](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/magistral) to start training your own Magistral models with Axolotl!
 - 2025/05: Quantization Aware Training (QAT) support has been added to Axolotl. Explore the [docs](https://docs.axolotl.ai/docs/qat.html) to learn more!
 - 2025/04: Llama 4 support has been added in Axolotl. See [examples](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/llama-4) to start training your own Llama 4 models with Axolotl's linearized version!
 - 2025/03: Axolotl has implemented Sequence Parallelism (SP) support. Read the [blog](https://huggingface.co/blog/axolotl-ai-co/long-context-with-sequence-parallelism-in-axolotl) and [docs](https://docs.axolotl.ai/docs/sequence_parallelism.html) to learn how to scale your context length when fine-tuning.
 - 2025/03: (Beta) Fine-tuning Multimodal models is now supported in Axolotl. Check out the [docs](https://docs.axolotl.ai/docs/multimodal.html) to fine-tune your own!
 - 2025/02: Axolotl has added LoRA optimizations to reduce memory usage and improve training speed for LoRA and QLoRA in single GPU and multi-GPU training (DDP and DeepSpeed). Jump into the [docs](https://docs.axolotl.ai/docs/lora_optims.html) to give it a try.
 - 2025/02: Axolotl has added GRPO support. Dive into our [blog](https://huggingface.co/blog/axolotl-ai-co/training-llms-w-interpreter-feedback-wasm) and [GRPO example](https://github.com/axolotl-ai-cloud/grpo_code) and have some fun!
 - 2025/01: Axolotl has added Reward Modelling / Process Reward Modelling fine-tuning support. See [docs](https://docs.axolotl.ai/docs/reward_modelling.html).
 </details>
 ## ✨ Overview
-Axolotl is a free and open-source tool designed to streamline post-training and fine-tuning for the latest large language models (LLMs).
+Axolotl is a tool designed to streamline post-training for various AI models.
 Features:
- **Multiple Model Support**: Train various models like GPT-OSS, LLaMA, Mistral, Mixtral, Pythia, and many more models available on the Hugging Face Hub.
+- **Multiple Model Support**: Train various models like LLaMA, Mistral, Mixtral, Pythia, and more. We are compatible with HuggingFace transformers causal language models.
- **Multimodal Training**: Fine-tune vision-language models (VLMs) including LLaMA-Vision, Qwen2-VL, Pixtral, LLaVA, SmolVLM2, and audio models like Voxtral with image, video, and audio support.
+- **Training Methods**: Full fine-tuning, LoRA, QLoRA, GPTQ, QAT, Preference Tuning (DPO, IPO, KTO, ORPO), RL (GRPO), Multimodal, and Reward Modelling (RM) / Process Reward Modelling (PRM).
- **Training Methods**: Full fine-tuning, LoRA, QLoRA, GPTQ, QAT, Preference Tuning (DPO, IPO, KTO, ORPO), RL (GRPO), and Reward Modelling (RM) / Process Reward Modelling (PRM).
+- **Easy Configuration**: Re-use a single YAML file between dataset preprocess, training, evaluation, quantization, and inference.
 - **Easy Configuration**: Re-use a single YAML configuration file across the full fine-tuning pipeline: dataset preprocessing, training, evaluation, quantization, and inference.
 - **Performance Optimizations**: [Multipacking](https://docs.axolotl.ai/docs/multipack.html), [Flash Attention](https://github.com/Dao-AILab/flash-attention), [Xformers](https://github.com/facebookresearch/xformers), [Flex Attention](https://pytorch.org/blog/flexattention/), [Liger Kernel](https://github.com/linkedin/Liger-Kernel), [Cut Cross Entropy](https://github.com/apple/ml-cross-entropy/tree/main), [Sequence Parallelism (SP)](https://docs.axolotl.ai/docs/sequence_parallelism.html), [LoRA optimizations](https://docs.axolotl.ai/docs/lora_optims.html), [Multi-GPU training (FSDP1, FSDP2, DeepSpeed)](https://docs.axolotl.ai/docs/multi-gpu.html), [Multi-node training (Torchrun, Ray)](https://docs.axolotl.ai/docs/multi-node.html), and many more!
 - **Flexible Dataset Handling**: Load from local, HuggingFace, and cloud (S3, Azure, GCP, OCI) datasets.
 - **Cloud Ready**: We ship [Docker images](https://hub.docker.com/u/axolotlai) and also [PyPI packages](https://pypi.org/project/axolotl/) for use on cloud platforms and local hardware.
-## 🚀 Quick Start - LLM Fine-tuning in Minutes
+## 🚀 Quick Start
 **Requirements**:
@@ -75,10 +57,6 @@ Features:
 - Python 3.11
 - PyTorch ≥2.6.0
 ### Google Colab
 [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/axolotl-ai-cloud/axolotl/blob/main/examples/colab-notebooks/colab-axolotl-example.ipynb#scrollTo=msOCO4NRmRLa)
 ### Installation
 #### Using pip
@@ -101,20 +79,6 @@ docker run --gpus '"all"' --rm -it axolotlai/axolotl:main-latest
 Other installation approaches are described [here](https://docs.axolotl.ai/docs/installation.html).
 #### Cloud Providers
 <details>
 - [RunPod](https://runpod.io/gsc?template=v2ickqhz9s&ref=6i7fkpdz)
 - [Vast.ai](https://cloud.vast.ai?ref_id=62897&template_id=bdd4a49fa8bce926defc99471864cace&utm_source=github&utm_medium=developer_community&utm_campaign=template_launch_axolotl&utm_content=readme)
 - [PRIME Intellect](https://app.primeintellect.ai/dashboard/create-cluster?image=axolotl&location=Cheapest&security=Cheapest&show_spot=true)
 - [Modal](https://www.modal.com?utm_source=github&utm_medium=github&utm_campaign=axolotl)
 - [Novita](https://novita.ai/gpus-console?templateId=311)
 - [JarvisLabs.ai](https://jarvislabs.ai/templates/axolotl)
 - [Latitude.sh](https://latitude.sh/blueprint/989e0e79-3bf6-41ea-a46b-1f246e309d5c)
 </details>
 ### Your First Fine-tune
 ```bash
@@ -156,22 +120,14 @@ Contributions are welcome! Please see our [Contributing Guide](https://github.co
 ## ❤️ Sponsors
 Thank you to our sponsors who help make Axolotl possible:
 - [Modal](https://www.modal.com?utm_source=github&utm_medium=github&utm_campaign=axolotl) - Modal lets you run
 jobs in the cloud, by just writing a few lines of Python. Customers use Modal to deploy Gen AI models at large scale,
 fine-tune large language models, run protein folding simulations, and much more.
 Interested in sponsoring? Contact us at [wing@axolotl.ai](mailto:wing@axolotl.ai)
 ## 📝 Citing Axolotl
 If you use Axolotl in your research or projects, please cite it as follows:
 ```bibtex
@software{axolotl,
  title = {Axolotl: Open Source LLM Post-Training},
  author = {{Axolotl maintainers and contributors}},
  url = {https://github.com/axolotl-ai-cloud/axolotl},
  license = {Apache-2.0},
  year = {2023}
 }
 ```
 ## 📜 License
 This project is licensed under the Apache 2.0 License - see the [LICENSE](LICENSE) file for details.
--- a/TODO.md
+++ b/TODO.md
@@ -0,0 +1,10 @@
 # todo list
 - [] Validation of parameters for combinations that won't work
 ## things that are known not to work
 - FSDP offload and gradient_checkpointing - https://github.com/pytorch/pytorch/issues/82203
 - adamw_bnb_8bit doesn't play well with FSDP offload
--- a/_quarto.yml
+++ b/_quarto.yml
@@ -35,30 +35,25 @@ quartodoc:
        - cli.train
        - cli.evaluate
        - cli.args
        - cli.art
        - cli.checks
        - cli.config
        - cli.delinearize_llama4
        - cli.inference
        - cli.merge_lora
        - cli.merge_sharded_fsdp_weights
        - cli.preprocess
-        - cli.quantize
+        - cli.sweeps
        - cli.utils
        - cli.vllm_serve
        - cli.cloud.base
        - cli.cloud.modal_
-        - cli.utils
+        - cli.quantize
        - cli.utils.args
        - cli.utils.fetch
        - cli.utils.load
        - cli.utils.sweeps
        - cli.utils.train
    - title: Trainers
      desc: Training implementations
      contents:
        - core.trainers.base
        - core.trainers.trl
        - core.trainers.mamba
        - core.trainers.relora
        - core.trainers.dpo.trainer
        - core.trainers.grpo.trainer
        - core.trainers.grpo.sampler
@@ -153,7 +148,7 @@ quartodoc:
        - utils.distributed
        - utils.dict
        - utils.optimizers.adopt
-        - utils.data.streaming
+        - utils.data.pretraining
        - utils.data.sft
        - utils.quantization
    - title: Schemas
@@ -272,10 +267,9 @@ website:
          contents:
            - docs/batch_vs_grad.qmd
            - docs/dataset_preprocessing.qmd
            - docs/streaming.qmd
            - docs/multipack.qmd
            - docs/mixed_precision.qmd
-            - docs/optimizers.qmd
+            - docs/gradient_accumulation.qmd
        - section: "Advanced Features"
          contents:
@@ -285,7 +279,6 @@ website:
            - docs/custom_integrations.qmd
            - docs/sequence_parallelism.qmd
            - docs/gradient_checkpointing.qmd
            - docs/nd_parallelism.qmd
        - section: "Troubleshooting"
          contents:
--- a/cicd/multigpu.py
+++ b/cicd/multigpu.py
@@ -2,6 +2,8 @@
 modal application to run axolotl gpu tests in Modal
 """
 # pylint: disable=duplicate-code
 import os
 import pathlib
 import tempfile
@@ -61,7 +63,7 @@ def run_cmd(cmd: str, run_folder: str):
    # Propagate errors from subprocess.
    if exit_code := subprocess.call(cmd.split(), cwd=run_folder):  # nosec
-        exit(exit_code)
+        exit(exit_code)  # pylint: disable=consider-using-sys-exit
@app.function(
--- a/cicd/multigpu.sh
+++ b/cicd/multigpu.sh
@@ -2,7 +2,7 @@
 set -e
 # Only run two tests at a time to avoid OOM on GPU (with coverage collection)
-pytest -v --durations=10 -n2 \
+pytest -v -n2 \
  --ignore=/workspace/axolotl/tests/e2e/multigpu/solo/ \
  --ignore=/workspace/axolotl/tests/e2e/multigpu/patched/ \
  /workspace/axolotl/tests/e2e/multigpu/ \
@@ -19,7 +19,5 @@ pytest -v  --durations=10 -n1 /workspace/axolotl/tests/e2e/multigpu/patched/ \
  --cov-append \
  --cov-report=xml:multigpu-coverage.xml
-# Upload coverage to Codecov if CODECOV_TOKEN is available
+# Upload coverage to Codecov
-if [ -n "$CODECOV_TOKEN" ]; then
+codecov upload-process -t "${CODECOV_TOKEN}" -f multigpu-coverage.xml -F multigpu,docker-tests,pytorch-${PYTORCH_VERSION} || true
  codecov upload-process -t "${CODECOV_TOKEN}" -f multigpu-coverage.xml -F multigpu,docker-tests,pytorch-${PYTORCH_VERSION} || true
 fi
--- a/cicd/single_gpu.py
+++ b/cicd/single_gpu.py
@@ -1,5 +1,7 @@
 """Modal app to run axolotl GPU tests"""
 # pylint: disable=duplicate-code
 import os
 import pathlib
 import tempfile
@@ -57,16 +59,12 @@ VOLUME_CONFIG = {
 }
 N_GPUS = int(os.environ.get("N_GPUS", 1))
-GPU_TYPE = os.environ.get("GPU_TYPE", "L40S")
+GPU_CONFIG = f"L40S:{N_GPUS}"
 GPU_CONFIG = f"{GPU_TYPE}:{N_GPUS}"
 def run_cmd(cmd: str, run_folder: str):
    import subprocess  # nosec
    sp_env = os.environ.copy()
    sp_env["AXOLOTL_DATASET_PROCESSES"] = "8"
    # Propagate errors from subprocess.
-    if exit_code := subprocess.call(cmd.split(), cwd=run_folder, env=sp_env):  # nosec
+    if exit_code := subprocess.call(cmd.split(), cwd=run_folder):  # nosec
-        exit(exit_code)
+        exit(exit_code)  # pylint: disable=consider-using-sys-exit
--- a/codecov.yml
+++ b/codecov.yml
@@ -12,7 +12,7 @@ coverage:
      default:
        # basic
        target: auto
-        threshold: 1%
+        threshold: 0%
        base: auto
        # advanced
        branches: null
@@ -27,7 +27,7 @@ coverage:
      default:
        # basic
        target: auto
-        threshold: 1%
+        threshold: 0%
        base: auto
        # advanced
        branches: null
--- a/docker/Dockerfile-base
+++ b/docker/Dockerfile-base
@@ -16,10 +16,7 @@ ENV PYTHON_VERSION=$PYTHON_VERSION
 ENV TORCH_CUDA_ARCH_LIST=$TORCH_CUDA_ARCH_LIST
 RUN apt-get update \
-    && apt-get install -y --no-install-recommends \
+    && apt-get install -y wget git build-essential ninja-build git-lfs libaio-dev pkg-config \
        wget git build-essential ninja-build git-lfs libaio-dev pkg-config \
        ibverbs-providers ibverbs-utils infiniband-diags  \
        librdmacm-dev librdmacm1 rdmacm-utils slurm-wlm \
    && rm -rf /var/cache/apt/archives \
    && rm -rf /var/lib/apt/lists/* \
    && wget \
@@ -37,7 +34,7 @@ WORKDIR /workspace
 RUN python3 -m pip install --upgrade pip && pip3 install -U packaging==23.2 setuptools==75.8.0 wheel && \
    python3 -m pip install --no-cache-dir -U torch==${PYTORCH_VERSION}+cu${CUDA} torchvision --extra-index-url https://download.pytorch.org/whl/cu$CUDA && \
-    CAUSAL_CONV1D_FORCE_CXX11_ABI=TRUE CAUSAL_CONV1D_FORCE_BUILD=TRUE python3 -m pip install --no-cache-dir causal_conv1d==1.5.2 && \
+    python3 -m pip install --no-cache-dir "causal_conv1d @ git+https://github.com/Dao-AILab/causal-conv1d.git@main" && \
    python3 -m pip install --no-cache-dir "mamba_ssm @ git+https://github.com/state-spaces/mamba.git@main" && \
    python3 -m pip cache purge
--- a/docker/Dockerfile-cloud
+++ b/docker/Dockerfile-cloud
@@ -15,7 +15,7 @@ COPY scripts/motd /etc/motd
 RUN pip install jupyterlab notebook ipywidgets && \
    jupyter lab clean
 RUN apt update && \
-    apt install --yes --no-install-recommends openssh-server tmux iproute2 nvtop && \
+    apt install --yes --no-install-recommends openssh-server tmux iproute2 nvtop ibverbs-providers ibverbs-utils infiniband-diags librdmacm-dev librdmacm1 rdmacm-utils slurm-wlm && \
    rm -rf /var/cache/apt/archives && \
    rm -rf /var/lib/apt/lists/* && \
    mkdir -p ~/.ssh && \
--- a/docker/Dockerfile-cloud-no-tmux
+++ b/docker/Dockerfile-cloud-no-tmux
@@ -9,15 +9,13 @@ ENV HF_HUB_ENABLE_HF_TRANSFER="1"
 EXPOSE 8888
 EXPOSE 22
-COPY scripts/cloud-entrypoint.sh /root/cloud-entrypoint.sh
+COPY scripts/cloud-entrypoint-term.sh /root/cloud-entrypoint.sh
 COPY scripts/motd /etc/motd
 RUN pip install jupyterlab notebook ipywidgets && \
    jupyter lab clean
-RUN apt update && \
+RUN apt install --yes --no-install-recommends openssh-server tmux sudo && \
-    apt install --yes --no-install-recommends openssh-server tmux iproute2 nvtop ibverbs-providers ibverbs-utils infiniband-diags librdmacm-dev librdmacm1 rdmacm-utils slurm-wlm && \
+    pip3 install -U --no-cache-dir grpcio ray[default]==2.9.3 && \
    rm -rf /var/cache/apt/archives && \
    rm -rf /var/lib/apt/lists/* && \
    mkdir -p ~/.ssh && \
    chmod 700 ~/.ssh && \
    printf "[ ! -z \"\$TERM\" -a -r /etc/motd ] && cat /etc/motd\n" >> ~/.bashrc && \
--- a/docs/cli.qmd
+++ b/docs/cli.qmd
@@ -23,20 +23,6 @@ axolotl <command> [config.yml] [options]
 The config file can be local or a URL to a raw YAML file.
 ### Launcher Arguments
 For commands that support multi-GPU (`train`, `evaluate`, ...), you can pass launcher-specific arguments using the `--` separator:
 ```bash
 # Pass torchrun arguments
 axolotl train config.yml --launcher torchrun -- --nproc_per_node=2 --nnodes=1
 # Pass accelerate arguments
 axolotl train config.yml --launcher accelerate -- --config_file=accelerate_config.yml --num_processes=4
 ```
 Arguments after `--` are passed directly to the launcher (torchrun, accelerate launch, etc.).
 ## Command Reference
 ### fetch
@@ -94,11 +80,7 @@ axolotl train config.yml \
    --num-epochs 3
 # Training without accelerate
-axolotl train config.yml --launcher python
+axolotl train config.yml --no-accelerate
 # Pass launcher-specific arguments using -- separator
 axolotl train config.yml --launcher torchrun -- --nproc_per_node=2 --nnodes=1
 axolotl train config.yml --launcher accelerate -- --config_file=accelerate_config.yml
 # Resume training from checkpoint
 axolotl train config.yml --resume-from-checkpoint path/to/checkpoint
@@ -193,9 +175,6 @@ Evaluates a model's performance (loss etc) on the train and eval datasets.
 ```bash
 # Basic evaluation
 axolotl evaluate config.yml
 # Evaluation with launcher arguments
 axolotl evaluate config.yml --launcher torchrun -- --nproc_per_node=2
 ```
 ### lm-eval
@@ -308,6 +287,9 @@ axolotl preprocess config.yml --cloud cloud_config.yml
 # Train on cloud
 axolotl train config.yml --cloud cloud_config.yml
 # Train without accelerate on cloud
 axolotl train config.yml --cloud cloud_config.yml --no-accelerate
 # Run lm-eval on cloud
 axolotl lm-eval config.yml --cloud cloud_config.yml
 ```
--- a/docs/dataset-formats/conversation.qmd
+++ b/docs/dataset-formats/conversation.qmd
@@ -212,11 +212,10 @@ Instead of passing `tools` via the system prompt, an alternative method would be
 Tools need to follow [JSON schema](https://json-schema.org/learn/getting-started-step-by-step).
 :::
 Example config for Llama4:
 ```yaml
 chat_template: llama4
 datasets:
-  - path: Nanobit/text-tools-2k-test
+  - path: ...
    type: chat_template
    # field_tools: tools # default is `tools`
 ```
--- a/docs/faq.qmd
+++ b/docs/faq.qmd
@@ -136,7 +136,3 @@ description: Frequently asked questions
 >   dynamic: false
 >   mode: max-autotune-no-cudagraphs
 > ```
 **Q: `ValueError("Backward pass should have cleared tracker of all tensors")`
 > A: This may happen due to edge cases in using the modern OffloadActivations context manager for CUDA streams. If you encounter this error, you may have success using the naive implementation with `offload_activations: legacy` in your YAML.
--- a/docs/installation.qmd
+++ b/docs/installation.qmd
@@ -124,17 +124,14 @@ For providers supporting Docker:
 - Use `axolotlai/axolotl-cloud:main-latest`
 - Available on:
-    - [RunPod](https://runpod.io/gsc?template=v2ickqhz9s&ref=6i7fkpdz)
+  - [Latitude.sh](https://latitude.sh/blueprint/989e0e79-3bf6-41ea-a46b-1f246e309d5c)
-    - [Vast.ai](https://cloud.vast.ai?ref_id=62897&template_id=bdd4a49fa8bce926defc99471864cace&utm_source=axolotl&utm_medium=partner&utm_campaign=template_launch_july2025&utm_content=docs_link)
+  - [JarvisLabs.ai](https://jarvislabs.ai/templates/axolotl)
-    - [PRIME Intellect](https://app.primeintellect.ai/dashboard/create-cluster?image=axolotl&location=Cheapest&security=Cheapest&show_spot=true)
+  - [RunPod](https://runpod.io/gsc?template=v2ickqhz9s&ref=6i7fkpdz)
-    - [Modal](https://www.modal.com?utm_source=github&utm_medium=github&utm_campaign=axolotl)
+  - [Novita](https://novita.ai/gpus-console?templateId=311)
    - [Novita](https://novita.ai/gpus-console?templateId=311)
    - [JarvisLabs.ai](https://jarvislabs.ai/templates/axolotl)
    - [Latitude.sh](https://latitude.sh/blueprint/989e0e79-3bf6-41ea-a46b-1f246e309d5c)
 ### Google Colab {#sec-colab}
-[![](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/axolotl-ai-cloud/axolotl/blob/main/examples/colab-notebooks/colab-axolotl-example.ipynb#scrollTo=msOCO4NRmRLa)
+Use our [example notebook](../examples/colab-notebooks/colab-axolotl-example.ipynb).
 ## Platform-Specific Instructions {#sec-platform-specific}
--- a/docs/multi-gpu.qmd
+++ b/docs/multi-gpu.qmd
@@ -63,6 +63,15 @@ Start from Stage 1 -> Stage 2 -> Stage 3.
 :::
 ::: {.callout-tip}
 Using ZeRO Stage 3 with Single-GPU training
 ZeRO Stage 3 can be used for training on a single GPU by manually setting the environment variables:
 `WORLD_SIZE=1 LOCAL_RANK=0 MASTER_ADDR=0.0.0.0 MASTER_PORT=29500`
 :::
 ## Fully Sharded Data Parallel (FSDP) {#sec-fsdp}
 ::: {.callout-note}
--- a/docs/multi-node.qmd
+++ b/docs/multi-node.qmd
@@ -69,19 +69,11 @@ export NCCL_BUFFSIZE=2097152
 Run the following on each node:
 ### Option 1: New Axolotl CLI with launcher args (Recommended)
 ```bash
 axolotl train config.yaml --launcher torchrun -- --nnodes $num_nodes --nproc_per_node $gpu_per_node --rdzv_id $rdzv_id --rdzv_backend c10d --rdzv_endpoint "$head_node_ip:$head_node_port"
 ```
 ### Option 2: Direct torchrun (Legacy)
 ```bash
 torchrun --nnodes $num_nodes --nproc_per_node $gpu_per_node --rdzv_id $rdzv_id --rdzv_backend c10d --rdzv_endpoint "$head_node_ip:$head_node_port" -m axolotl.cli.train config.yaml
 ```
-Please make sure to substitute the placeholder variables:
+Please make sure to substitute the placeholder variables.
 - `num_nodes`: Number of nodes (containing GPUs)
 - `gpu_per_node`: Number of gpus per node
@@ -89,6 +81,8 @@ Please make sure to substitute the placeholder variables:
 - `head_node_port`: Port of the head node (make sure other machines can connect to this. Default 29400)
 - `rdzv_id`: A unique job ID that is used by the job across nodes.
-The new CLI approach (Option 1) is recommended as it provides consistent argument handling and works seamlessly with other Axolotl CLI features.
+::: {.callout-note}
 You need to call `axolotl.cli.train` instead of `axolotl train` as the latter calls accelerate under the hood
 :::
 More info on the available configs can be found on the Pytorch docs [here](https://pytorch.org/docs/stable/elastic/run.html)
--- a/docs/multimodal.qmd
+++ b/docs/multimodal.qmd
@@ -13,13 +13,10 @@ format:
 - [Pixtral](#sec-pixtral)
 - [Llava-1.5](#sec-llava-15)
 - [Mistral-Small-3.1](#sec-mistral-small-31)
 - [Voxtral](#sec-voxtral)
 - [Gemma-3](#sec-gemma-3)
 - [Gemma-3n](#sec-gemma-3n)
 - [Qwen2-VL](#sec-qwen2-vl)
 - [Qwen2.5-VL](#sec-qwen25-vl)
 - [SmolVLM2](#sec-smolvlm2)
 - [LFM2-VL](#sec-lfm2-vl)
 ## Usage
@@ -34,7 +31,7 @@ skip_prepare_dataset: true
 remove_unused_columns: false  # leave columns in place as they are needed to handle image embeddings during training
 sample_packing: false  # not yet supported with multimodal
-chat_template:  # see in next section if specified
+chat_template:  # see in next section
 # example dataset
 datasets:
@@ -100,16 +97,6 @@ base_model: mistralai/Mistral-Small-3.1-24B-Instruct-2503
 chat_template: mistral_v7_tekken
 ```
 ### Voxtral {#sec-voxtral}
 ::: {.callout-tip}
 Please make sure to install audio lib via `pip3 install librosa==0.11.0 'mistral_common[audio]==1.8.3'`
 :::
 ```yaml
 base_model: mistralai/Voxtral-Mini-3B-2507
 ```
 ### Gemma-3 {#sec-gemma-3}
 ::: {.callout-tip}
@@ -156,26 +143,6 @@ base_model: Qwen/Qwen2.5-VL-7B-Instruct
 chat_template: qwen2_vl  # same as qwen2-vl
 ```
 ### SmolVLM2 {#sec-smolvlm2}
 ::: {.callout-tip}
 Please make sure to install `num2words` via `pip3 install num2words==0.5.14`
 :::
 ```yaml
 base_model: HuggingFaceTB/SmolVLM2-500M-Video-Instruct
 ```
 ### LFM2-VL {#sec-lfm2-vl}
 ::: {.callout-warning}
 Please uninstall `causal-conv1d` via `pip3 uninstall -y causal-conv1d`
 :::
 ```yaml
 base_model: LiquidAI/LFM2-VL-450M
 ```
 ## Dataset Format
 For multi-modal datasets, we adopt an extended `chat_template` format similar to OpenAI's Message format.
@@ -214,20 +181,6 @@ You may need to install `librosa` via `pip3 install librosa==0.11.0`.
 :::
 ### Video
 ::: {.callout-warning}
 This is not well tested at the moment. We welcome contributors!
 :::
 For video loading, you can use the following keys within `content` alongside `"type": "video"`:
 - `"path": "/path/to/video.mp4"`
 - `"url": "https://example.com/video.mp4"`
 - `"video": np.ndarray | list[PIL.Image.Image] | torch.Tensor` (or list of the aforementioned)
 ### Example
 Here is an example of a multi-modal dataset:
--- a/docs/nd_parallelism.qmd
+++ b/docs/nd_parallelism.qmd
@@ -1,108 +0,0 @@
 ---
 title: "N-D Parallelism (Beta)"
 ---
 Axolotl enables training models at scale by composing different parallelism techniques. This is essential when:
 - A model's weights are too large to fit on a single GPU's memory.
 - A model's activations, especially with very long contexts, are too large for a single GPU.
 - You want to accelerate training by using multiple GPUs or nodes.
 or combinations of the above!
 ## Core Concepts
 Parallelism strategies can be combined. The key is understanding how each one divides the workload. PyTorch's `DeviceMesh` is the modern way to manage these combinations, creating a logical grid of your GPUs and assigning different parallel strategies to different dimensions of the grid.
 ### Data Parallelism {#sec-dp}
 Data Parallelism focuses on splitting the global data batch across GPUs.
 - Distributed Data Parallel (DDP): The classic approach. The full model is replicated on every GPU. Each GPU processes a different slice of the data batch. Gradients are then averaged across all GPUs after the backward pass to keep the models synchronized. This can substantially improve data throughput compared to single-device training, but requires that each GPU is able to hold the entire model, its gradients, and optimizer states.
 - [Fully Sharded Data Parallel (FSDP)](multi-gpu.qmd#fully-sharded-data-parallel-(fsdp)): A highly memory-efficient form of data parallelism (inspired by DeepSpeed's ZeRO). Instead of replicating the model, FSDP shards the model's *parameters, gradients, and optimizer states* across the GPUs in the data-parallel group. During computation, each GPU receives the specific parameters it needs via an `all_gather` operation just before they are used, and they can be discarded immediately after (`reshard-after-forward`).
    - FSDP maps to ZeRO stages:
        - ZeRO-2 (`reshard_after_forward=False`): Shards gradients and optimizer states. Model weights are replicated on each GPU.
        - ZeRO-3 (`reshard_after_forward=True`): Shards gradients, optimizer states, AND model parameters. This provides the most memory savings at the cost of more communication (re-gathering parameters for both forward and backward passes).
 ### [Experimental] Tensor Parallelism (TP) {#sec-tp}
 Also known as "horizontal model parallelism," as described in the [Megatron-LM paper](https://arxiv.org/pdf/1909.08053.pdf). Instead of splitting the batch, TP splits the model's layers themselves across GPUs.
 - How it works: For a linear layer `Y = XA`, the weight matrix `A` is split column-wise (`A = [A_1, A_2]`). The computation becomes `Y_1 = XA_1` and `Y_2 = XA_2`, which can happen in parallel on different GPUs. The final output `Y` is simply the concatenation of `Y_1` and `Y_2`. Check [this comment](https://github.com/huggingface/transformers/issues/10321#issuecomment-783543530) for more detailed info.
 - Requirement: TP involves frequent, small communications within a forward/backward pass. It requires a very fast interconnect between GPUs (e.g., NVLink) and is typically not recommended across different nodes.
 ### Context Parallelism (CP) {#sec-cp}
 Context Parallelism, also called [Sequence Parallelism](sequence_parallelism.qmd), addresses the memory bottleneck from long sequences. The input sequence itself is split along the sequence length dimension and distributed across GPUs.
 - How it works: If you have a sequence of 8192 tokens and a `context_parallel_size` of 4, each GPU will only handle a chunk of 2048 tokens.
 - The Challenge: Attention is not local; every token needs to "attend to" every other token. Splitting the sequence breaks this.
 - The Solution (`ring-flash-attention`): An efficient communication protocol is used. To compute attention for its local sequence chunk, each GPU passes its Key-Value (KV) cache to its neighbor in a "ring." After `N-1` steps, every GPU has seen the KV-cache from all other GPUs, allowing it to compute the correct attention values for its chunk. This is implemented using the highly optimized `flash-attention` kernel at each step.
 ### Hybrid Sharding Data Parallel (HSDP) {#sec-hsdp}
 HSDP is a 2D strategy that intelligently combines FSDP and DDP, typically for multi-node training.
 - Intra-Node (within a machine): Use FSDP. This is efficient because GPUs on the same node have fast interconnects (NVLink), making the `all_gather` operations for sharded parameters fast.
 - Inter-Node (across machines): Use DDP. The gradient synchronization between nodes is less frequent than FSDP's parameter gathering, making it a better fit for the slower node-to-node network (e.g., Ethernet/Infiniband).
 - Example: With 2 nodes of 8 GPUs each (16 total), you could have `dp_shard_size=8` (FSDP within each node) and `dp_replicate_size=2` (DDP across the two nodes).
 ## Usage
 ```yaml
 # FSDP config. See https://docs.axolotl.ai/docs/multi-gpu.html#sec-fsdp
 fsdp_version: 2
 fsdp_config:
  # ...
 # The number of GPUs to shard the model parameters across (FSDP dimension).
 dp_shard_size: 4
 # The number of times to replicate the sharded model (DDP dimension).
 dp_replicate_size: 2
 # Number of GPUs for Tensor Parallelism.
 tensor_parallel_size: 1  # (default is 1, no TP)
 # Number of GPUs for Context/Sequence Parallelism.
 context_parallel_size: 1 # (default is 1, no CP)
 ```
 Note: We recommend FSDP. DeepSpeed is only compatible with `tensor_parallel_size`.
 ## Examples
 ::: {.callout-tip}
 See our example configs [here](https://github.com/axolotl-ai-cloud/axolotl/tree/main/examples/distributed-parallel).
 :::
 1.  HSDP on 2 nodes with 4 GPUs each (8 GPUs total):
    - You want FSDP within each node and DDP across nodes.
    - Set `dp_shard_size: 4` and `dp_replicate_size: 2`.
 2.  FSDP + TP on a single 8-GPU node:
    - You want to split the model across 4 GPUs using FSDP, and further split each layer across 2 GPUs with TP.
    - Set `dp_shard_size: 4` and `tensor_parallel_size: 2`.
 3.  FSDP + CP on a single 8-GPU node for long context:
    - You want to shard the model across all 8 GPUs and also split the sequence length across all 8 GPUs.
    - Set `dp_shard_size: 8` and `context_parallel_size: 8`. Note: this means the data parallel group and context parallel group are the same. A more common setup might be to shard across a smaller group.
 ## Support Matrix
 This matrix describes how different parallelism methods can be combined in Axolotl.
 | Combination | `dp_replicate_size` | `dp_shard_size` | `tp_size` | `cp_size` | Status & Notes |
 | --- | :---: | :---: |:---:|:---:|---|
 | **FSDP** (ZeRO-3) | 1 | >1 | 1 | 1 | ✅ Fully supported. Shards model across all GPUs. |
 | **HSDP** | >1 | >1 | 1 | 1 | ✅ Fully supported. FSDP intra-node, DDP inter-node. |
 | **FSDP + TP** | 1 | >1 | >1 | 1 | ✅ **2D Parallelism**. Shards the model across a `dp_shard` group, and TP-splits layers within the `tp` group. |
 | **HSDP + TP** | >1 | >1 | >1 | 1 | ✅ **3D Parallelism**. A powerful but complex combination. |
 | **FSDP + CP** | 1 | >1 | 1 | >1 | ✅ **2D Parallelism**. Combines FSDP with context parallelism. |
 | **FSDP + TP + CP**| 1 | >1 | >1| >1| ✅ **3D Parallelism**. Another advanced combination. |
 | DDP + TP/CP | >1 | 1 | >1 | >1 | ❌ **Not Supported**. The `ParallelismConfig` explicitly prevents this, as composing pure DDP with TP or CP is currently not supported. You should use FSDP + TP/CP instead (`dp_shard_size > 1`). |
 | Just TP / CP | 1 | 1 | >1 | >1 | ✅ Supported. Useful for inference or when the model fits on one GPU but context is too long. |
 - `tp_size` refers to `tensor_parallel_size`
 - `cp_size` refers to `context_parallel_size`
--- a/docs/optimizers.qmd
+++ b/docs/optimizers.qmd
@@ -1,129 +0,0 @@
 ---
 title: Optimizers
 description: Configuring optimizers
 ---
 ## Overview
 Axolotl supports all optimizers supported by [transformers OptimizerNames](https://github.com/huggingface/transformers/blob/51f94ea06d19a6308c61bbb4dc97c40aabd12bad/src/transformers/training_args.py#L142-L187)
 Here is a list of optimizers supported by transformers as of `v4.54.0`:
 - `adamw_torch`
 - `adamw_torch_fused`
 - `adamw_torch_xla`
 - `adamw_torch_npu_fused`
 - `adamw_apex_fused`
 - `adafactor`
 - `adamw_anyprecision`
 - `adamw_torch_4bit`
 - `adamw_torch_8bit`
 - `ademamix`
 - `sgd`
 - `adagrad`
 - `adamw_bnb_8bit`
 - `adamw_8bit`  # alias for adamw_bnb_8bit
 - `ademamix_8bit`
 - `lion_8bit`
 - `lion_32bit`
 - `paged_adamw_32bit`
 - `paged_adamw_8bit`
 - `paged_ademamix_32bit`
 - `paged_ademamix_8bit`
 - `paged_lion_32bit`
 - `paged_lion_8bit`
 - `rmsprop`
 - `rmsprop_bnb`
 - `rmsprop_bnb_8bit`
 - `rmsprop_bnb_32bit`
 - `galore_adamw`
 - `galore_adamw_8bit`
 - `galore_adafactor`
 - `galore_adamw_layerwise`
 - `galore_adamw_8bit_layerwise`
 - `galore_adafactor_layerwise`
 - `lomo`
 - `adalomo`
 - `grokadamw`
 - `schedule_free_radam`
 - `schedule_free_adamw`
 - `schedule_free_sgd`
 - `apollo_adamw`
 - `apollo_adamw_layerwise`
 - `stable_adamw`
 ## Custom Optimizers
 Enable custom optimizers by passing a string to the `optimizer` argument. Each optimizer will receive beta and epsilon args, however, some may accept additional args which are detailed below.
 ### optimi_adamw
 ```yaml
 optimizer: optimi_adamw
 ```
 ### ao_adamw_4bit
 Deprecated: Please use `adamw_torch_4bit`.
 ### ao_adamw_8bit
 Deprecated: Please use `adamw_torch_8bit`.
 ### ao_adamw_fp8
 ```yaml
 optimizer: ao_adamw_fp8
 ```
 ### adopt_adamw
 GitHub: [https://github.com/iShohei220/adopt](https://github.com/iShohei220/adopt)
 Paper: [https://arxiv.org/abs/2411.02853](https://arxiv.org/abs/2411.02853)
 ```yaml
 optimizer: adopt_adamw
 ```
 ### came_pytorch
 GitHub: [https://github.com/yangluo7/CAME/tree/master](https://github.com/yangluo7/CAME/tree/master)
 Paper: [https://arxiv.org/abs/2307.02047](https://arxiv.org/abs/2307.02047)
 ```yaml
 optimizer: came_pytorch
 # optional args (defaults below)
 adam_beta1: 0.9
 adam_beta2: 0.999
 adam_beta3: 0.9999
 adam_epsilon: 1e-30
 adam_epsilon2: 1e-16
 ```
 ### muon
 Blog: [https://kellerjordan.github.io/posts/muon/](https://kellerjordan.github.io/posts/muon/)
 Paper: [https://arxiv.org/abs/2502.16982v1](https://arxiv.org/abs/2502.16982v1)
 ```yaml
 optimizer: muon
 ```
 ### dion
 Microsoft's Dion (DIstributed OrthoNormalization) optimizer is a scalable and communication-efficient
 orthonormalizing optimizer that uses low-rank approximations to reduce gradient communication.
 GitHub: [https://github.com/microsoft/dion](https://github.com/microsoft/dion)
 Paper: [https://arxiv.org/pdf/2504.05295](https://arxiv.org/pdf/2504.05295)
 Note: Implementation written for PyTorch 2.7+ for DTensor
 ```yaml
 optimizer: dion
 dion_lr: 0.01
 dion_momentum: 0.95
 lr: 0.00001  # learning rate for embeddings and parameters that fallback to AdamW
 ```
--- a/docs/quantize.qmd
+++ b/docs/quantize.qmd
@@ -51,11 +51,3 @@ axolotl quantize qat.yml
 ```
 This ensures that an identical quantization configuration is used to quantize the model as was used to train it.
 ::: {.callout-note}
 If you have configured pushing to hub with `hub_model_id`, your model hub name will have the quantization schema appended to it,
 e.g. `axolotl-ai-cloud/qat-nvfp4-llama3B` will become `axolotl-ai-cloud/qat-nvfp4-llama3B-nvfp4w`
 :::
--- a/docs/reward_modelling.qmd
+++ b/docs/reward_modelling.qmd
@@ -11,7 +11,6 @@ We support the reward modelling techniques supported by `trl`.
 ### (Outcome) Reward Models
 Outcome reward models are trained using data which contains preference annotations for an entire interaction between the user and model (e.g. rather than per-turn or per-step).
 For improved training stability, you can use the `center_rewards_coefficient` parameter to encourage mean-zero reward outputs ([see TRL docs](https://huggingface.co/docs/trl/v0.10.1/en/reward_trainer#centering-rewards)).
 ```yaml
 base_model: google/gemma-2-2b
--- a/docs/scripts/generate_config_docs.py
+++ b/docs/scripts/generate_config_docs.py
@@ -47,6 +47,7 @@ class QuartoGenerator:
        """Check if a type is a Pydantic BaseModel."""
        return inspect.isclass(type_obj) and issubclass(type_obj, BaseModel)
    # pylint: disable=too-many-return-statements
    def _extract_nested_type(self, field_type) -> Any:
        """Extract the actual type from complex type annotations."""
        # Handle Annotated types (Python 3.9+)
@@ -123,6 +124,7 @@ class QuartoGenerator:
        return field_type
    # pylint: disable=too-many-return-statements
    def _extract_all_pydantic_models_from_type(
        self, field_type
    ) -> list[type[BaseModel]]:
@@ -316,6 +318,7 @@ class QuartoGenerator:
        return all_groups
    # pylint: disable=too-many-return-statements
    def _extract_field_groups_from_source(
        self, model_class: type[BaseModel]
    ) -> list[dict]:
@@ -500,7 +503,7 @@ class QuartoGenerator:
                    nested_schema = nested_model.model_json_schema()
                    nested_properties = nested_schema.get("properties", {})
                    nested_required = nested_schema.get("required", [])
-                except Exception:
+                except Exception:  # pylint: disable=broad-exception-caught
                    # Fallback: use model fields directly
                    nested_properties = {}
                    nested_required = []
@@ -604,7 +607,7 @@ class QuartoGenerator:
            schema = model_class.model_json_schema()
            properties = schema.get("properties", {})
            required = schema.get("required", [])
-        except Exception as e:
+        except Exception as e:  # pylint: disable=broad-exception-caught
            print(
                f"Warning: Could not generate JSON schema ({e}). Using model fields instead."
            )
--- a/docs/streaming.qmd
+++ b/docs/streaming.qmd
@@ -1,120 +0,0 @@
 ---
 title: Streaming Datasets
 description: How to use streaming mode for large-scale datasets and memory-efficient training
 order: 10
 ---
 Streaming enables memory-efficient training with large datasets by loading data
 incrementally rather than loading the entire dataset into memory at once.
 Use streaming when:
 - Your dataset is too large to fit in memory (e.g. when you're doing pretraining with massive text corpora)
 - You want to start training immediately without preprocessing the entire dataset
 Streaming works with both remote and locally stored datasets!
 ::: {.callout-note}
 Streaming currently only supports a single dataset. Multi-dataset support will be added soon.
 :::
 ## Configuration
 ### Basic Streaming
 Enable streaming mode by setting the `streaming` flag:
 ```yaml
 streaming: true
 ```
 ### Pretraining with Streaming
 For pretraining tasks, streaming is automatically enabled when using `pretraining_dataset`:
 ```yaml
 pretraining_dataset:
  - path: HuggingFaceFW/fineweb-edu
    type: pretrain
    text_column: text
    split: train
 # Optionally, enable sample packing
 streaming_multipack_buffer_size: 10000
 sample_packing: true
 ```
 ### SFT with Streaming
 For supervised fine-tuning with streaming:
 ```yaml
 streaming: true
 datasets:
  - path: tatsu-lab/alpaca
    type: alpaca
    split: train
 # Optionally, enable sample packing
 streaming_multipack_buffer_size: 10000
 sample_packing: true
 ```
 ## Configuration Options
 ### `streaming_multipack_buffer_size`
 Controls the buffer size for multipack streaming (default: 10,000). This determines how
 many samples are buffered before packing. Larger buffers can improve packing efficiency
 but use more memory.
 ### `shuffle_merged_datasets`
 When enabled, shuffles the streaming dataset using the buffer. This requires additional
 memory for the shuffle buffer.
 ## Sample Packing with Streaming
 Sample packing is supported for streaming datasets. When enabled, multiple samples are
 packed into a single sequence to maximize GPU utilization:
 ```yaml
 sample_packing: true
 streaming_multipack_buffer_size: 10000
 # For SFT: attention is automatically isolated between packed samples
 # For pretraining: control with pretrain_multipack_attn
 pretrain_multipack_attn: true  # prevent cross-attention between packed samples
 ```
 For more information, see our [documentation](multipack.qmd) on multipacking.
 ## Important Considerations
 ### Memory Usage
 While streaming reduces memory usage compared to loading entire datasets, you still need
 to consider:
 - You can control the memory usage by adjusting `streaming_multipack_buffer_size`
 - Sample packing requires buffering multiple samples
 - Shuffling requires additional memory for the shuffle buffer
 ### Performance
 - Streaming may have slightly higher latency compared to preprocessed datasets, as samples are processed on-the-fly
 - Network speed and disk read speed are important when streaming from remote sources or a local dataset, respectively
 - Consider using `axolotl preprocess` for smaller or more frequently used datasets
 ### Evaluation Datasets
 Evaluation datasets are not streamed to ensure consistent evaluation metrics. They're
 loaded normally even when training uses streaming.
 ## Examples
 See the `examples/streaming/` directory for complete configuration examples:
 - `pretrain.yaml`: Pretraining with streaming dataset
 - `sft.yaml`: Supervised fine-tuning with streaming
--- a/examples/LiquidAI/README.md
+++ b/examples/LiquidAI/README.md
@@ -1,58 +0,0 @@
 # Finetune Liquid Foundation Models 2 (LFM2) with Axolotl
 [Liquid Foundation Models 2 (LFM2)](https://huggingface.co/collections/LiquidAI/lfm2-686d721927015b2ad73eaa38) are a family of small, open-weight models from [Liquid AI](https://www.liquid.ai/) focused on quality, speed, and memory efficiency. Liquid AI released text-only [LFM2](https://huggingface.co/collections/LiquidAI/lfm2-686d721927015b2ad73eaa38) and text+vision [LFM2-VL](https://huggingface.co/collections/LiquidAI/lfm2-vl-68963bbc84a610f7638d5ffa) models.
 LFM2 features a new hybrid Liquid architecture with multiplicative gates, short-range convolutions, and grouped query attention, enabling fast training and inference.
 This guide shows how to fine-tune both the LFM2 and LFM2-VL models with Axolotl.
 ## Getting Started
 1.  Install Axolotl following the [installation guide](https://docs.axolotl.ai/docs/installation.html).
    Here is an example of how to install from pip:
    ```bash
    # Ensure you have a compatible version of Pytorch installed
    pip3 install packaging setuptools wheel ninja
    pip3 install --no-build-isolation 'axolotl[flash-attn]>=0.12.0'
    ```
 2.  Run one of the finetuning examples below.
    **LFM2**
    ```bash
    # FFT SFT (1x48GB @ 25GiB)
    axolotl train examples/LiquidAI/lfm2-350m-fft.yaml
    ```
    **LFM2-VL**
    ```bash
    # LoRA SFT (1x48GB @ 2.7GiB)
    axolotl train examples/LiquidAI/lfm2-vl-lora.yaml
    ```
 ### TIPS
 - **Installation Error**: If you encounter `ImportError: ... undefined symbol ...` or `ModuleNotFoundError: No module named 'causal_conv1d_cuda'`, the `causal-conv1d` package may have been installed incorrectly. Try uninstalling it:
  ```bash
  pip uninstall -y causal-conv1d
  ```
 - **Dataset Loading**: Read more on how to load your own dataset in our [documentation](https://docs.axolotl.ai/docs/dataset_loading.html).
 - **Dataset Formats**:
  - For LFM2 models, the dataset format follows the OpenAI Messages format as seen [here](https://docs.axolotl.ai/docs/dataset-formats/conversation.html#chat_template).
  - For LFM2-VL models, Axolotl follows the multi-content Messages format. See our [Multimodal docs](https://docs.axolotl.ai/docs/multimodal.html#dataset-format) for details.
 ## Optimization Guides
 - [Multi-GPU Training](https://docs.axolotl.ai/docs/multi-gpu.html)
 - [LoRA Optimizations](https://docs.axolotl.ai/docs/lora_optims.html)
 - [Multi-Node Training](https://docs.axolotl.ai/docs/multi-node.html)
 ## Related Resources
 - [LFM2 Blog](https://www.liquid.ai/blog/liquid-foundation-models-v2-our-second-series-of-generative-ai-models)
 - [LFM2-VL Blog](https://www.liquid.ai/blog/lfm2-vl-efficient-vision-language-models)
 - [Axolotl Docs](https://docs.axolotl.ai)
 - [Axolotl GitHub](https://github.com/axolotl-ai-cloud/axolotl)
 - [Axolotl Discord](https://discord.gg/7m9sfhzaf3)
--- a/examples/LiquidAI/lfm2-vl-lora.yaml
+++ b/examples/LiquidAI/lfm2-vl-lora.yaml
@@ -1,58 +0,0 @@
 base_model: LiquidAI/LFM2-VL-450M
 trust_remote_code: true
 model_type: AutoModelForImageTextToText
 processor_type: AutoProcessor
 # these 3 lines are needed for now to handle vision chat templates w images
 skip_prepare_dataset: true
 remove_unused_columns: false
 sample_packing: false
 datasets:
  - path: HuggingFaceH4/llava-instruct-mix-vsft
    type: chat_template
    split: train[:1%]
 dataset_prepared_path: last_run_prepared
 val_set_size: 0.0
 output_dir: ./outputs/out
 adapter: lora
 lora_model_dir:
 sequence_len: 8192
 pad_to_sequence_len: false
 lora_r: 32
 lora_alpha: 16
 lora_dropout: 0.05
 lora_target_modules: 'model.language_model.layers.[\d]+.(mlp|cross_attn|self_attn).(up|down|gate|q|k|v|o)_proj'
 wandb_project:
 wandb_entity:
 wandb_watch:
 wandb_name:
 wandb_log_model:
 gradient_accumulation_steps: 4
 micro_batch_size: 1
 num_epochs: 1
 optimizer: adamw_bnb_8bit
 lr_scheduler: cosine
 learning_rate: 0.0002
 bf16: true
 fp16:
 tf32: true
 gradient_checkpointing: true
 logging_steps: 1
 flash_attention: true
 eager_attention:
 warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
 # save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/alst/README.md
+++ b/examples/alst/README.md
@@ -1,9 +0,0 @@
 # Arctic Long Sequence Training (ALST)
 Artic Long Sequence Training (ALST) is a technique for training long context models using a variety of optimization
 techniques. It is a combination of:
 - TiledMLP: Leverage tiling over the sequence dimension on MLP layers to reduce memory usage
 - Tiled Loss: Using optimized loss functions like Liger-Kernel or Cut Cross Entropy to reduce memory usage
 - Activation Offloading: Offload activations to CPU RAM to reduce memory usage
 For more information, you can check out the ALST paper [here](https://www.arxiv.org/abs/2506.13996).
--- a/examples/alst/llama3-8b-deepspeed-alst.yaml
+++ b/examples/alst/llama3-8b-deepspeed-alst.yaml
@@ -1,53 +0,0 @@
 base_model: meta-llama/Llama-3.1-8B
 # Automatically upload checkpoint and final model to HF
 # hub_model_id: username/custom_model_name
 datasets:
  - path: togethercomputer/Long-Data-Collections
    type: completion
    field: text
    data_files:
      - pretrain/rp_sub.jsonl.zst
  - path: princeton-nlp/TextbookChapters
    type: completion
    field: chapter
 dataset_prepared_path: last_run_prepared
 val_set_size: 0.0
 output_dir: ./outputs/out
 sequence_len: 500_000
 min_sample_len: 200_000
 sample_packing: true
 tiled_mlp: true
 context_parallel_size: 8
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 gradient_accumulation_steps: 1
 micro_batch_size: 1
 num_epochs: 1
 optimizer: adamw_torch_8bit
 lr_scheduler: cosine
 learning_rate: 2e-5
 bf16: auto
 tf32: true
 gradient_checkpointing: true
 activation_offloading: legacy
 resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
 warmup_steps: 100
 saves_per_epoch: 1
 evals_per_epoch: 2
 weight_decay: 0.0
 special_tokens:
  pad_token: <|end_of_text|>
 deepspeed: deepspeed_configs/zero3_bf16_cpuoffload_all.json
 # save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/alst/llama3-8b-fsdp2-alst.yaml
+++ b/examples/alst/llama3-8b-fsdp2-alst.yaml
@@ -1,59 +0,0 @@
 base_model: meta-llama/Llama-3.1-8B
 # Automatically upload checkpoint and final model to HF
 # hub_model_id: username/custom_model_name
 datasets:
  - path: togethercomputer/Long-Data-Collections
    type: completion
    field: text
    data_files:
      - pretrain/rp_sub.jsonl.zst
  - path: princeton-nlp/TextbookChapters
    type: completion
    field: chapter
 dataset_prepared_path: last_run_prepared
 val_set_size: 0.0
 output_dir: ./outputs/out
 sequence_len: 500_000
 min_sample_len: 200_000
 sample_packing: true
 tiled_mlp: true
 context_parallel_size: 8
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 gradient_accumulation_steps: 1
 micro_batch_size: 1
 num_epochs: 1
 optimizer: adamw_torch_8bit
 lr_scheduler: cosine
 learning_rate: 2e-5
 bf16: auto
 tf32: true
 gradient_checkpointing: true
 activation_offloading: legacy
 resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
 warmup_steps: 100
 saves_per_epoch: 1
 evals_per_epoch: 2
 weight_decay: 0.0
 special_tokens:
  pad_token: <|end_of_text|>
 fsdp_version: 2
 fsdp_config:
  offload_params: false  # offloading is currently not compatible with SP + torchao optimizer
  state_dict_type: SHARDED_STATE_DICT
  auto_wrap_policy: TRANSFORMER_BASED_WRAP
  transformer_layer_cls_to_wrap: LlamaDecoderLayer
  reshard_after_forward: true
 # save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/arcee/README.md
+++ b/examples/arcee/README.md
@@ -1,53 +0,0 @@
 # Finetune ArceeAI's AFM with Axolotl
 [Arcee Foundation Models (AFM)](https://huggingface.co/collections/arcee-ai/afm-45b-68823397c351603014963473) are a family of 4.5B parameter open weight models trained by Arcee.ai.
 This guide shows how to fine-tune it with Axolotl with multi-turn conversations and proper masking.
 Thanks to the team at Arcee.ai for using Axolotl in supervised fine-tuning the AFM model.
 ## Getting started
 1. Install Axolotl following the [installation guide](https://docs.axolotl.ai/docs/installation.html). You need to install from main as AFM is only on nightly or use our latest [Docker images](https://docs.axolotl.ai/docs/docker.html).
    Here is an example of how to install from main for pip:
 ```bash
 # Ensure you have Pytorch installed (Pytorch 2.6.0 min)
 git clone https://github.com/axolotl-ai-cloud/axolotl.git
 cd axolotl
 pip3 install packaging==23.2 setuptools==75.8.0 wheel ninja
 pip3 install --no-build-isolation -e '.[flash-attn]'
 ```
 2. Run the finetuning example:
 ```bash
 axolotl train examples/arcee/afm-4.5b-qlora.yaml
 ```
 This config uses about 7.8GiB VRAM.
 Let us know how it goes. Happy finetuning! 🚀
 ### TIPS
 - For inference, the official Arcee.ai team recommends `top_p: 0.95`, `temperature: 0.5`, `top_k: 50`, and `repeat_penalty: 1.1`.
 - You can run a full finetuning by removing the `adapter: qlora` and `load_in_4bit: true` from the config.
 - Read more on how to load your own dataset at [docs](https://docs.axolotl.ai/docs/dataset_loading.html).
 - The dataset format follows the OpenAI Messages format as seen [here](https://docs.axolotl.ai/docs/dataset-formats/conversation.html#chat_template).
 ## Optimization Guides
 - [Multi-GPU Training](https://docs.axolotl.ai/docs/multi-gpu.html)
 - [Multi-Node Training](https://docs.axolotl.ai/docs/multi-node.html)
 - [LoRA Optimizations](https://docs.axolotl.ai/docs/lora_optims.html)
 ## Related Resources
 - [AFM Blog](https://docs.arcee.ai/arcee-foundation-models/introduction-to-arcee-foundation-models)
 - [Axolotl Docs](https://docs.axolotl.ai)
 - [Axolotl Website](https://axolotl.ai)
 - [Axolotl GitHub](https://github.com/axolotl-ai-cloud/axolotl)
 - [Axolotl Discord](https://discord.gg/7m9sfhzaf3)
--- a/examples/arcee/afm-4.5b-qlora.yaml
+++ b/examples/arcee/afm-4.5b-qlora.yaml
@@ -1,64 +0,0 @@
 base_model: arcee-ai/AFM-4.5B
 # Automatically upload checkpoint and final model to HF
 # hub_model_id: username/custom_model_name
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 load_in_8bit: false
 load_in_4bit: true
 datasets:
  - path: fozziethebeat/alpaca_messages_2k_test
    type: chat_template
 dataset_prepared_path: last_run_prepared
 val_set_size: 0.1
 output_dir: ./outputs/lora-out
 adapter: qlora
 lora_model_dir:
 sequence_len: 2048
 sample_packing: true
 lora_r: 32
 lora_alpha: 16
 lora_dropout: 0.05
 lora_target_linear: true
 lora_target_modules:
  - gate_proj
  - down_proj
  - up_proj
  - q_proj
  - v_proj
  - k_proj
  - o_proj
 wandb_project:
 wandb_entity:
 wandb_watch:
 wandb_name:
 wandb_log_model:
 gradient_accumulation_steps: 4
 micro_batch_size: 2
 num_epochs: 1
 optimizer: adamw_bnb_8bit
 lr_scheduler: cosine
 learning_rate: 0.0002
 bf16: auto
 tf32: false
 gradient_checkpointing: true
 resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
 warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 # save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/archived/cerebras/btlm-ft.yml
+++ b/examples/archived/cerebras/btlm-ft.yml
@@ -66,7 +66,7 @@ flash_optimum:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 32
 evals_per_epoch: 4
 saves_per_epoch: 1
 save_total_limit:
--- a/examples/archived/cerebras/qlora.yml
+++ b/examples/archived/cerebras/qlora.yml
@@ -43,7 +43,7 @@ xformers_attention: true
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.1
--- a/examples/archived/code-llama/13b/lora.yml
+++ b/examples/archived/code-llama/13b/lora.yml
@@ -47,7 +47,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/code-llama/13b/qlora.yml
+++ b/examples/archived/code-llama/13b/qlora.yml
@@ -48,7 +48,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/code-llama/34b/lora.yml
+++ b/examples/archived/code-llama/34b/lora.yml
@@ -47,7 +47,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/code-llama/34b/qlora.yml
+++ b/examples/archived/code-llama/34b/qlora.yml
@@ -48,7 +48,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/code-llama/7b/lora.yml
+++ b/examples/archived/code-llama/7b/lora.yml
@@ -47,7 +47,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/code-llama/7b/qlora.yml
+++ b/examples/archived/code-llama/7b/qlora.yml
@@ -48,7 +48,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/dbrx/16bit-lora.yaml
+++ b/examples/archived/dbrx/16bit-lora.yaml
@@ -54,7 +54,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch:
 saves_per_epoch: 1
--- a/examples/archived/dbrx/8bit-lora.yaml
+++ b/examples/archived/dbrx/8bit-lora.yaml
@@ -57,7 +57,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch:
 saves_per_epoch: 1
--- a/examples/archived/dbrx/fft-ds-zero3.yaml
+++ b/examples/archived/dbrx/fft-ds-zero3.yaml
@@ -41,7 +41,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch:
 saves_per_epoch: 1
--- a/examples/archived/deepcoder/deepcoder-14B-preview-lora.yml
+++ b/examples/archived/deepcoder/deepcoder-14B-preview-lora.yml
@@ -51,7 +51,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/falcon/config-7b-lora.yml
+++ b/examples/archived/falcon/config-7b-lora.yml
@@ -47,7 +47,7 @@ xformers_attention: true
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 40
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/falcon/config-7b-qlora.yml
+++ b/examples/archived/falcon/config-7b-qlora.yml
@@ -77,7 +77,7 @@ xformers_attention: true
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.000001
--- a/examples/archived/falcon/config-7b.yml
+++ b/examples/archived/falcon/config-7b.yml
@@ -44,7 +44,7 @@ xformers_attention: true
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 40
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/gptj/qlora.yml
+++ b/examples/archived/gptj/qlora.yml
@@ -40,7 +40,7 @@ xformers_attention: true
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.1
--- a/examples/archived/jeopardy-bot/config.yml
+++ b/examples/archived/jeopardy-bot/config.yml
@@ -41,7 +41,7 @@ xformers_attention: true
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.1
--- a/examples/archived/mpt-7b/config.yml
+++ b/examples/archived/mpt-7b/config.yml
@@ -42,7 +42,7 @@ logging_steps: 5
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0001
--- a/examples/archived/openllama-3b/config.yml
+++ b/examples/archived/openllama-3b/config.yml
@@ -42,7 +42,7 @@ logging_steps: 1
 flash_attention: true
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.1
--- a/examples/archived/openllama-3b/lora.yml
+++ b/examples/archived/openllama-3b/lora.yml
@@ -50,7 +50,7 @@ logging_steps: 1
 flash_attention: true
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.1
--- a/examples/archived/openllama-3b/qlora.yml
+++ b/examples/archived/openllama-3b/qlora.yml
@@ -43,7 +43,7 @@ logging_steps: 1
 flash_attention: true
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.1
--- a/examples/archived/qwen/lora.yml
+++ b/examples/archived/qwen/lora.yml
@@ -49,7 +49,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention:
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/qwen/qlora.yml
+++ b/examples/archived/qwen/qlora.yml
@@ -49,7 +49,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention:
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/qwen/qwen2-moe-lora.yaml
+++ b/examples/archived/qwen/qwen2-moe-lora.yaml
@@ -45,7 +45,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/qwen/qwen2-moe-qlora.yaml
+++ b/examples/archived/qwen/qwen2-moe-qlora.yaml
@@ -48,7 +48,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/redpajama/config-3b.yml
+++ b/examples/archived/redpajama/config-3b.yml
@@ -43,7 +43,7 @@ logging_steps: 5
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0001
--- a/examples/archived/replit-3b/config-lora.yml
+++ b/examples/archived/replit-3b/config-lora.yml
@@ -41,7 +41,7 @@ logging_steps: 1
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0
--- a/examples/archived/stablelm-2/1.6b/fft.yml
+++ b/examples/archived/stablelm-2/1.6b/fft.yml
@@ -47,9 +47,10 @@ logging_steps: 1
 flash_attention: true
 flash_attn_cross_entropy: false
 flash_attn_rms_norm: true
 flash_attn_fuse_qkv: false
 flash_attn_fuse_mlp: true
-warmup_ratio: 0.1
+warmup_steps: 100
 evals_per_epoch: 4
 saves_per_epoch: 1
--- a/examples/archived/stablelm-2/1.6b/lora.yml
+++ b/examples/archived/stablelm-2/1.6b/lora.yml
@@ -51,7 +51,7 @@ flash_attention: true
 flash_attn_cross_entropy: false
 flash_attn_rms_norm: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/starcoder2/qlora.yml
+++ b/examples/archived/starcoder2/qlora.yml
@@ -48,7 +48,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 20
 evals_per_epoch: 4
 eval_steps:
 saves_per_epoch: 4
--- a/examples/archived/tiny-llama/lora-mps.yml
+++ b/examples/archived/tiny-llama/lora-mps.yml
@@ -49,7 +49,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: false
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 0
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/tiny-llama/lora.yml
+++ b/examples/archived/tiny-llama/lora.yml
@@ -47,7 +47,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/tiny-llama/pretrain.yml
+++ b/examples/archived/tiny-llama/pretrain.yml
@@ -38,7 +38,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/tiny-llama/qlora.yml
+++ b/examples/archived/tiny-llama/qlora.yml
@@ -49,7 +49,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/xgen-7b/xgen-7b-8k-qlora.yml
+++ b/examples/archived/xgen-7b/xgen-7b-8k-qlora.yml
@@ -75,7 +75,7 @@ xformers_attention: true
 flash_attention:
 gptq_groupsize:
 gptq_model_v1:
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/archived/yi-34B-chat/qlora.yml
+++ b/examples/archived/yi-34B-chat/qlora.yml
@@ -20,7 +20,7 @@ special_tokens:
 datasets:
  - path: mhenrichsen/alpaca_2k_test
    type: alpaca
-warmup_ratio: 0.1
+warmup_steps: 10
 # Iterations
 num_epochs: 1
--- a/examples/cloud/baseten.yaml
+++ b/examples/cloud/baseten.yaml
@@ -1,10 +0,0 @@
 provider: baseten
 project_name:
 secrets:
  - HF_TOKEN
  - WANDB_API_KEY
 gpu: h100
 gpu_count: 8
 node_count: 1
--- a/examples/colab-notebooks/colab-axolotl-example.ipynb
+++ b/examples/colab-notebooks/colab-axolotl-example.ipynb
--- a/examples/deepcogito/cogito-v1-preview-llama-3B-lora.yml
+++ b/examples/deepcogito/cogito-v1-preview-llama-3B-lora.yml
@@ -51,7 +51,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/deepcogito/cogito-v1-preview-qwen-14B-lora.yml
+++ b/examples/deepcogito/cogito-v1-preview-qwen-14B-lora.yml
@@ -51,7 +51,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/deepseek-v2/fft-fsdp-16b.yaml
+++ b/examples/deepseek-v2/fft-fsdp-16b.yaml
@@ -37,7 +37,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 100
 evals_per_epoch: 2
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/deepseek-v2/qlora-fsdp-2_5.yaml
+++ b/examples/deepseek-v2/qlora-fsdp-2_5.yaml
@@ -61,7 +61,7 @@ resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
-warmup_ratio: 0.1
+warmup_steps: 100
 evals_per_epoch: 2
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/devstral/README.md
+++ b/examples/devstral/README.md
@@ -10,23 +10,20 @@ Thanks to the team at MistralAI for giving us early access to prepare for this r
 ## Getting started
-1. Install Axolotl following the [installation guide](https://docs.axolotl.ai/docs/installation.html).
+1. Install Axolotl following the [installation guide](https://docs.axolotl.ai/docs/installation.html). You need to install from main as Devstral is only on nightly or use our latest [Docker images](https://docs.axolotl.ai/docs/docker.html).
-    Here is an example of how to install from pip:
+    Here is an example of how to install from main for pip:
 ```bash
-# Ensure you have Pytorch installed (Pytorch 2.6.0 min)
+# Ensure you have Pytorch installed (Pytorch 2.6.0+)
 git clone https://github.com/axolotl-ai-cloud/axolotl.git
 cd axolotl
 pip3 install packaging==23.2 setuptools==75.8.0 wheel ninja
-pip3 install --no-build-isolation 'axolotl[flash-attn]>=0.12.0'
+pip3 install --no-build-isolation -e '.[flash-attn]'
 ```
-2. Install [Cut Cross Entropy](https://docs.axolotl.ai/docs/custom_integrations.html#cut-cross-entropy) to reduce training VRAM usage
+2. Run the finetuning example:
 ```bash
 python scripts/cutcrossentropy_install.py | sh
 ```
 3. Run the finetuning example:
 ```bash
 axolotl train examples/devstral/devstral-small-qlora.yml
--- a/examples/distributed-parallel/README.md
+++ b/examples/distributed-parallel/README.md
@@ -1,52 +0,0 @@
 # ND Parallelism Examples
 This directory contains example configurations for training models using ND Parallelism in Axolotl. These examples demonstrate how to compose different parallelism strategies (FSDP, TP, CP, HSDP) for efficient multi-GPU training.
 ## Quick Start
 1. Install Axolotl following the [installation guide](https://docs.axolotl.ai/docs/installation.html).
 2. Run the command below:
 ```bash
 # Train Qwen3 8B with FSDP + TP + CP on a single 8-GPU node
 axolotl train examples/distributed-parallel/qwen3-8b-fsdp-tp-cp.yaml
 # Train Llama 3.1 8B with HSDP + TP on 2 nodes (16 GPUs total)
 axolotl train examples/distributed-parallel/llama-3_1-8b-hsdp-tp.yaml
 ```
 ## Example Configurations
 ### Single Node (8 GPUs)
 **Qwen3 8B with FSDP + TP + CP** ([qwen3-8b-fsdp-tp-cp.yaml](./qwen3-8b-fsdp-tp-cp.yaml))
 - Uses all 3 parallelism dimensions on a single node
 - Ideal for: when model weights, activations, and/or context are too large to fit on single GPU
 ```yaml
 dp_shard_size: 2         # FSDP across 2 GPUs
 tensor_parallel_size: 2  # TP across 2 GPUs
 context_parallel_size: 2 # CP across 2 GPUs
 # Total: 2 × 2 × 2 = 8 GPUs
 ```
 ### Multi-Node
 **Llama 3.1 8B with HSDP + TP** ([llama-3_1-8b-hsdp-tp.yaml](./llama-3_1-8b-hsdp-tp.yaml))
 - FSDP & TP within nodes, DDP across nodes to minimize inter-node communication
 - Ideal for: Scaling to multiple nodes while maintaining training efficiency
 ```yaml
 dp_shard_size: 4        # FSDP within each 4-GPU group
 tensor_parallel_size: 2 # TP within each node
 dp_replicate_size: 2    # DDP across 2 groups
 # Total: (4 × 2) × 2 = 16 GPUs (2 nodes)
 ```
 ## Learn More
 - [ND Parallelism Documentation](https://docs.axolotl.ai/docs/nd_parallelism.html)
 - [Blog: Accelerate ND-Parallel Guide](https://huggingface.co/blog/accelerate-nd-parallel)
 - [Multi-GPU Training Guide](https://docs.axolotl.ai/docs/multi-gpu.html)
 - [Axolotl Discord](https://discord.gg/7m9sfhzaf3)
--- a/examples/distributed-parallel/llama-3_1-8b-hsdp-tp.yaml
+++ b/examples/distributed-parallel/llama-3_1-8b-hsdp-tp.yaml
@@ -1,47 +0,0 @@
 base_model: meta-llama/Llama-3.1-8B
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 dp_shard_size: 4
 dp_replicate_size: 2
 tensor_parallel_size: 2
 # context_parallel_size: 2
 dataset_prepared_path: last_run_prepared
 special_tokens:
  pad_token: <|end_of_text|>
 fsdp_version: 2
 fsdp_config:
  offload_params: false
  state_dict_type: FULL_STATE_DICT
  auto_wrap_policy: TRANSFORMER_BASED_WRAP
  transformer_layer_cls_to_wrap: LlamaDecoderLayer
  reshard_after_forward: true
 datasets:
  - path: tatsu-lab/alpaca
    type: alpaca
 output_dir: ./outputs/ndp-out/
 sequence_len: 2048
 sample_packing: true
 flash_attention: true
 gradient_accumulation_steps: 1
 micro_batch_size: 1
 num_epochs: 2
 optimizer: adamw_torch_fused
 lr_scheduler: constant_with_warmup
 learning_rate: 2e-6
 bf16: true
 tf32: true
 logging_steps: 1
 saves_per_epoch: 1
 warmup_ratio: 0.1
--- a/examples/distributed-parallel/qwen3-8b-fsdp-tp-cp.yaml
+++ b/examples/distributed-parallel/qwen3-8b-fsdp-tp-cp.yaml
@@ -1,46 +0,0 @@
 base_model: Qwen/Qwen3-8B
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 dp_shard_size: 2
 # dp_replicate_size: 1
 context_parallel_size: 2
 tensor_parallel_size: 2
 dataset_prepared_path: last_run_prepared
 fsdp_version: 2
 fsdp_config:
  offload_params: false
  state_dict_type: FULL_STATE_DICT
  auto_wrap_policy: TRANSFORMER_BASED_WRAP
  transformer_layer_cls_to_wrap: Qwen3DecoderLayer
  reshard_after_forward: true
 datasets:
  - path: tatsu-lab/alpaca
    type: alpaca
 output_dir: ./outputs/ndp-out/
 sequence_len: 8192
 sample_packing: true
 flash_attention: true
 gradient_accumulation_steps: 1
 micro_batch_size: 1  # must be 1 when using context parallel
 num_epochs: 2
 optimizer: adamw_torch_fused
 lr_scheduler: constant_with_warmup
 learning_rate: 2e-6
 bf16: true
 tf32: true
 logging_steps: 1
 saves_per_epoch: 1
 warmup_ratio: 0.1
 special_tokens:
--- a/examples/gemma3/270m-qlora.yml
+++ b/examples/gemma3/270m-qlora.yml
@@ -1,68 +0,0 @@
 base_model: google/gemma-3-270m-it
 # optionally might have model_type or tokenizer_type
 model_type: AutoModelForCausalLM
 tokenizer_type: AutoTokenizer
 # Automatically upload checkpoint and final model to HF
 # hub_model_id: username/custom_model_name
 # gemma3 doesn't seem to play nice with ddp
 ddp_find_unused_parameters: true
 load_in_8bit: false
 load_in_4bit: true
 # huggingface repo
 chat_template: gemma3
 eot_tokens:
  - <end_of_turn>
 datasets:
  - path: cgato/SlimOrcaDedupCleaned
    type: chat_template
    field_messages: conversations
    message_property_mappings:
      role: from
      content: value
 val_set_size: 0.0
 output_dir: ./outputs/out
 adapter: qlora
 lora_r: 32
 lora_alpha: 16
 lora_dropout: 0.05
 lora_target_linear: true
 sequence_len: 2048
 sample_packing: true
 eval_sample_packing: false
 wandb_project:
 wandb_entity:
 wandb_watch:
 wandb_name:
 wandb_log_model:
 gradient_accumulation_steps: 4
 micro_batch_size: 1
 num_epochs: 1
 optimizer: adamw_bnb_8bit
 lr_scheduler: cosine
 learning_rate: 0.0002
 bf16: auto
 tf32: true
 gradient_checkpointing: true
 gradient_checkpointing_kwargs:
  use_reentrant: false
 resume_from_checkpoint:
 logging_steps: 1
 flash_attention: true
 warmup_ratio: 0.1
 evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
--- a/examples/gemma3n/README.md
+++ b/examples/gemma3n/README.md
@@ -1,62 +1,19 @@
-# Finetune Gemma-3n with Axolotl
+# Gemma-3n
-Gemma-3n is a family of multimodal models from Google found on [HuggingFace](https://huggingface.co/collections/google/gemma-3n-685065323f5984ef315c93f4). This guide shows how to fine-tune it with Axolotl.
+## Requirements
-## Getting started
+In addition to Axolotl's requirements, Gemma-3n requires
-1. Install Axolotl following the [installation guide](https://docs.axolotl.ai/docs/installation.html).
+```
-
+pip3 install timm
    Here is an example of how to install from pip:
 ```bash
 # Ensure you have Pytorch installed (Pytorch 2.6.0 min)
 pip3 install packaging==23.2 setuptools==75.8.0 wheel ninja
 pip3 install --no-build-isolation 'axolotl[flash-attn]>=0.12.0'
 ```
-2. In addition to Axolotl's requirements, Gemma-3n requires:
+If you will load audio datasets, please also install
-```bash
+```
-pip3 install timm==1.0.17
+pip3 install librosa
 # for loading audio data
 pip3 install librosa==0.11.0
 ```
-3. Run the finetuning example:
+## Usage
-```bash
+See example configs and the [multimodal doc](https://docs.axolotl.ai/docs/multimodal.html).
 # text only
 axolotl train examples/gemma3n/gemma-3n-e2b-qlora.yml
 # text + vision
 axolotl train examples/gemma3n/gemma-3n-e2b-vision-qlora.yml
 # text + vision + audio
 axolotl train examples/gemma3n/gemma-3n-e2b-vision-audio-qlora.yml
 ```
 Let us know how it goes. Happy finetuning! 🚀
 WARNING: The loss and grad norm will be much higher than normal. We suspect this to be inherent to the model as of the moment. If anyone would like to submit a fix for this, we are happy to take a look.
 ### TIPS
 - You can run a full finetuning by removing the `adapter: qlora` and `load_in_4bit: true` from the config.
 - Read more on how to load your own dataset at [docs](https://docs.axolotl.ai/docs/dataset_loading.html).
 - The text dataset format follows the OpenAI Messages format as seen [here](https://docs.axolotl.ai/docs/dataset-formats/conversation.html#chat_template).
 - The multimodal dataset format follows the OpenAI multi-content Messages format as seen [here](https://docs.axolotl.ai/docs/multimodal.html#dataset-format).
 ## Optimization Guides
 - [Multi-GPU Training](https://docs.axolotl.ai/docs/multi-gpu.html)
 - [Multi-Node Training](https://docs.axolotl.ai/docs/multi-node.html)
 - [LoRA Optimizations](https://docs.axolotl.ai/docs/lora_optims.html)
 ## Related Resources
 - [Gemma 3n Blog](https://ai.google.dev/gemma/docs/gemma-3n)
 - [Axolotl Docs](https://docs.axolotl.ai)
 - [Axolotl Website](https://axolotl.ai)
 - [Axolotl GitHub](https://github.com/axolotl-ai-cloud/axolotl)
 - [Axolotl Discord](https://discord.gg/7m9sfhzaf3)
--- a/examples/gemma3n/gemma-3n-e2b-vision-audio-qlora.yml
+++ b/examples/gemma3n/gemma-3n-e2b-vision-audio-qlora.yml
@@ -34,6 +34,8 @@ eot_tokens:
 datasets:
  - path: Nanobit/text-vision-audio-2k-test
    type: chat_template
    data_files:
      - dataset.jsonl
 dataset_prepared_path:
 val_set_size: 0.01
 output_dir: ./outputs/out
--- a/examples/glm4/qlora-32b.yaml
+++ b/examples/glm4/qlora-32b.yaml
@@ -55,7 +55,7 @@ flash_attention: true
 loss_watchdog_threshold: 5.0
 loss_watchdog_patience: 3
-warmup_ratio: 0.1
+warmup_steps: 10
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
--- a/examples/gpt-oss/README.md
+++ b/examples/gpt-oss/README.md
@@ -1,135 +0,0 @@
 # Finetune OpenAI's GPT-OSS with Axolotl
 [GPT-OSS](https://huggingface.co/collections/openai/gpt-oss-68911959590a1634ba11c7a4) are a family of open-weight MoE models trained by OpenAI, released in August 2025. There are two variants: 20B and 120B.
 This guide shows how to fine-tune it with Axolotl with multi-turn conversations and proper masking.
 ## Getting started
 1. Install Axolotl following the [installation guide](https://docs.axolotl.ai/docs/installation.html).
    Here is an example of how to install from pip:
 ```bash
 # Ensure you have Pytorch installed (Pytorch 2.6.0 min)
 pip3 install packaging==23.2 setuptools==75.8.0 wheel ninja
 pip3 install --no-build-isolation 'axolotl[flash-attn]>=0.12.0'
 ```
 2. Choose one of the following configs below for training the 20B model. (for 120B, see [below](#training-120b))
 ```bash
 # LoRA SFT linear layers (1x48GB @ ~44GiB)
 axolotl train examples/gpt-oss/gpt-oss-20b-sft-lora-singlegpu.yaml
 # FFT SFT with offloading (2x24GB @ ~21GiB/GPU)
 axolotl train examples/gpt-oss/gpt-oss-20b-fft-fsdp2-offload.yaml
 # FFT SFT (8x48GB @ ~36GiB/GPU or 4x80GB @ ~46GiB/GPU)
 axolotl train examples/gpt-oss/gpt-oss-20b-fft-fsdp2.yaml
 ```
 Note: Memory usage taken from `device_mem_reserved(gib)` from logs.
 ### Training 120B
 On 8xH100s, make sure you have ~3TB of free disk space. With each checkpoint clocking in at ~720GB, along with the base
 model, and final model output, you may need at least 3TB of free disk space to keep at least 2 checkpoints.
 ```bash
 # FFT SFT with offloading (8x80GB @ ~49GiB/GPU)
 axolotl train examples/gpt-oss/gpt-oss-120b-fft-fsdp2-offload.yaml
 ```
 To simplify fine-tuning across 2 nodes × 8x H100 (80GB) GPUs, we've partnered with [Baseten](https://baseten.co) to showcase multi-node
 training of the 120B model using Baseten Truss. You can read more about this recipe on
 [Baseten's blog](https://www.baseten.co/blog/how-to-fine-tune-gpt-oss-120b-with-baseten-and-axolotl/). The recipe can
 be found on their
 [GitHub](https://github.com/basetenlabs/ml-cookbook/tree/main/examples/oss-gpt-120b-axolotl/training).
 ERRATA: Transformers saves the model Architecture prefixed with `FSDP` which needs to be manually renamed in `config.json`.
 See https://github.com/huggingface/transformers/pull/40207 for the status of this issue.
 ```bash
 sed -i 's/FSDPGptOssForCausalLM/GptOssForCausalLM/g' ./outputs/gpt-oss-out/config.json
 ```
 When using SHARDED_STATE_DICT with FSDP, the final checkpoint should automatically merge the sharded weights to your
 configured `output_dir`. However, if that step fails due to a disk space error, you can take an additional step to
 merge the sharded weights.  This step will automatically determine the last checkpoint directory and merge the sharded
 weights to `{output_dir}/merged`.
 ```bash
 axolotl merge-sharded-fsdp-weights examples/gpt-oss/gpt-oss-120b-fft-fsdp2-offload.yaml
 mv ./outputs/gpt-oss-out/merged/* ./outputs/gpt-oss-out/
 ```
 ### Inferencing your fine-tuned model
 #### vLLM
 GPT-OSS support in vLLM does not exist in a stable release yet. See https://x.com/MaziyarPanahi/status/1955741905515323425
 for more information about using a special vllm-openai docker image for inferencing with vLLM.
 Optionally, vLLM can be installed from nightly:
 ```bash
 pip install --no-build-isolation --pre -U vllm --extra-index-url https://wheels.vllm.ai/nightly
 ```
 and the vLLM server can be started with the following command (modify `--tensor-parallel-size 8` to match your environment):
 ```bash
 vllm serve ./outputs/gpt-oss-out/ --served-model-name axolotl/gpt-oss-20b --host 0.0.0.0 --port 8888  --tensor-parallel-size 8
 ```
 #### SGLang
 SGLang has 0-day support in main, see https://github.com/sgl-project/sglang/issues/8833 for infomation on installing
 SGLang from source. Once you've installed SGLang, run the following command to launch a SGLang server:
 ```bash
 python3 -m sglang.launch_server --model ./outputs/gpt-oss-out/ --served-model-name axolotl/gpt-oss-120b --host 0.0.0.0 --port 8888 --tp 8
 ```
 ### Tool use
 GPT-OSS has a comprehensive tool understanding. Axolotl supports tool calling datasets for Supervised Fine-tuning.
 Here is an example dataset config:
 ```yaml
 datasets:
  - path: Nanobit/text-tools-2k-test
    type: chat_template
 ```
 See [Nanobit/text-tools-2k-test](https://huggingface.co/datasets/Nanobit/text-tools-2k-test) for the sample dataset.
 Refer to [our docs](https://docs.axolotl.ai/docs/dataset-formats/conversation.html#using-tool-use) for more info.
 ### Thinking and chat_template masking conflict
 OpenAI’s Harmony template hides `thinking` in all non-final turns, which conflicts with Axolotl’s `chat_template` masking.
 If your dataset has `thinking` content mid-turn, there are two paths we recommend:
 - Train only on the last turn. This can be accomplished via chat_template's [train on last doc](https://docs.axolotl.ai/docs/dataset-formats/conversation.html#training-on-last-message).
 - Adjust your dataset to only have `thinking` content in the last turn.
 ### TIPS
 - Read more on how to load your own dataset at [docs](https://docs.axolotl.ai/docs/dataset_loading.html).
 - The dataset format follows the OpenAI Messages format as seen [here](https://docs.axolotl.ai/docs/dataset-formats/conversation.html#chat_template).
 ## Optimization Guides
 - [Multi-GPU Training](https://docs.axolotl.ai/docs/multi-gpu.html)
 - [Multi-Node Training](https://docs.axolotl.ai/docs/multi-node.html)
 ## Related Resources
 - [GPT-OSS Blog](https://openai.com/index/introducing-gpt-oss/)
 - [Axolotl Docs](https://docs.axolotl.ai)
 - [Axolotl Website](https://axolotl.ai)
 - [Axolotl GitHub](https://github.com/axolotl-ai-cloud/axolotl)
 - [Axolotl Discord](https://discord.gg/7m9sfhzaf3)
--- a/examples/gpt-oss/gpt-oss-120b-fft-fsdp2-offload.yaml
+++ b/examples/gpt-oss/gpt-oss-120b-fft-fsdp2-offload.yaml
@@ -1,68 +0,0 @@
 # the original mxfp4 quantized model is not supported with FSDP cpu_ram_efficient_loading
 # FSDP cpu_ram_efficient_loading is used to reduce the initial CPU memory usage when loading the model
 base_model: axolotl-ai-co/gpt-oss-120b-dequantized
 use_kernels: false
 dp_shard_size: 16  # requires 2x8xH100 nodes
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 experimental_skip_move_to_device: true  # prevent OOM by NOT putting model to GPU before sharding
 datasets:
  - path: HuggingFaceH4/Multilingual-Thinking
    type: chat_template
    field_thinking: thinking
    template_thinking_key: thinking
 dataset_prepared_path: last_run_prepared
 val_set_size: 0
 output_dir: ./outputs/gpt-oss-out/
 save_total_limit: 2  # the 120B model can use up to 720GB of disk space per checkpoint, so let's only keep the last 2
 sequence_len: 4096
 sample_packing: true
 pad_to_sequence_len: true
 wandb_project:
 wandb_entity:
 wandb_watch:
 wandb_name:
 wandb_log_model:
 gradient_accumulation_steps: 2
 micro_batch_size: 1
 num_epochs: 1
 optimizer: adamw_torch_fused  # 8bit optimizers do not work with FSDP2 offload
 lr_scheduler: constant_with_warmup
 learning_rate: 2e-5
 bf16: true
 tf32: true
 flash_attention: true
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3
 gradient_checkpointing: true
 activation_offloading: true
 logging_steps: 1
 saves_per_epoch: 1
 warmup_ratio: 0.03
 special_tokens:
 eot_tokens:
  - "<|end|>"
 fsdp_version: 2
 fsdp_config:
  offload_params: true
  state_dict_type: SHARDED_STATE_DICT
  auto_wrap_policy: TRANSFORMER_BASED_WRAP
  transformer_layer_cls_to_wrap: GptOssDecoderLayer
  reshard_after_forward: true
  cpu_ram_efficient_loading: true
--- a/examples/gpt-oss/gpt-oss-20b-fft-deepspeed-zero3.yaml
+++ b/examples/gpt-oss/gpt-oss-20b-fft-deepspeed-zero3.yaml
@@ -1,58 +0,0 @@
 base_model: openai/gpt-oss-20b
 use_kernels: false
 model_quantization_config: Mxfp4Config
 model_quantization_config_kwargs:
  dequantize: true
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 experimental_skip_move_to_device: true  # prevent OOM by NOT putting model to GPU before sharding
 datasets:
  - path: HuggingFaceH4/Multilingual-Thinking
    type: chat_template
    field_thinking: thinking
    template_thinking_key: thinking
 dataset_prepared_path: last_run_prepared
 val_set_size: 0
 output_dir: ./outputs/gpt-oss-out/
 sequence_len: 4096
 sample_packing: true
 wandb_project:
 wandb_entity:
 wandb_watch:
 wandb_name:
 wandb_log_model:
 gradient_accumulation_steps: 2
 micro_batch_size: 1
 num_epochs: 1
 optimizer: adamw_torch_8bit
 lr_scheduler: constant_with_warmup
 learning_rate: 2e-5
 bf16: true
 tf32: true
 flash_attention: true
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3
 gradient_checkpointing: true
 activation_offloading: true
 logging_steps: 1
 saves_per_epoch: 1
 warmup_ratio: 0.03
 special_tokens:
 eot_tokens:
  - "<|end|>"
 # choose the zero3 configuration that best fits your system capabilities
 deepspeed: deepspeed_configs/zero3_bf16.json
--- a/examples/gpt-oss/gpt-oss-20b-fft-fsdp2-offload.yaml
+++ b/examples/gpt-oss/gpt-oss-20b-fft-fsdp2-offload.yaml
@@ -1,68 +0,0 @@
 base_model: openai/gpt-oss-20b
 use_kernels: true
 model_quantization_config: Mxfp4Config
 model_quantization_config_kwargs:
  dequantize: true
 plugins:
  - axolotl.integrations.cut_cross_entropy.CutCrossEntropyPlugin
 experimental_skip_move_to_device: true  # prevent OOM by NOT putting model to GPU before sharding
 datasets:
  - path: HuggingFaceH4/Multilingual-Thinking
    type: chat_template
    field_thinking: thinking
    template_thinking_key: thinking
 dataset_prepared_path: ./outputs/last_run_prepared
 val_set_size: 0
 output_dir: ./outputs/gpt-oss-out/
 sequence_len: 4096
 sample_packing: true
 pad_to_sequence_len: true
 wandb_project:
 wandb_entity:
 wandb_watch:
 wandb_name:
 wandb_log_model:
 gradient_accumulation_steps: 2
 micro_batch_size: 1
 num_epochs: 1
 optimizer: adamw_torch_fused  # 8bit optimizers do not work with FSDP2 offload
 lr_scheduler: constant_with_warmup
 learning_rate: 2e-5
 bf16: true
 tf32: true
 flash_attention: true
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3
 gradient_checkpointing: true
 activation_offloading: true
 logging_steps: 1
 saves_per_epoch: 1
 warmup_ratio: 0.03
 special_tokens:
 eot_tokens:
  - "<|end|>"
 fsdp_version: 2
 fsdp_config:
  offload_params: true
  state_dict_type: SHARDED_STATE_DICT
  auto_wrap_policy: TRANSFORMER_BASED_WRAP
  transformer_layer_cls_to_wrap: GptOssDecoderLayer
  reshard_after_forward: true
  #  cpu_ram_efficient_loading: true
 # cpu_ram_efficient_loading cannot be used with MXFP4 model quantization.
 # It can only be used with a dequantized model like `axolotl-ai-co/gpt-oss-120b-dequantized`
--- a/Show More
+++ b/Show More
Author	SHA1	Message	Date
Salman Mohammadi	bc2bc688d8	update fsdp2 patch	2025-07-23 16:53:03 +01:00
Wing Lian	b3c04dd9fe	workaround for fsdp2 optimizer save failures	2025-07-23 09:38:57 -04:00
Wing Lian	972c719d38	use latest transformers on main with fix	2025-07-23 09:22:36 -04:00
Wing Lian	2c1cb8b300	fix for accelerator state getting reset and missing schema	2025-07-23 08:43:34 -04:00
Wing Lian	cca207eec4	handle none checks	2025-07-22 21:21:45 -04:00
Wing Lian	9a2da4d9f0	update tp validation	2025-07-22 21:20:57 -04:00
Wing Lian	8fe4758e94	make sure to return data for validation	2025-07-22 21:18:39 -04:00
Wing Lian	8c641fdcb4	handle tp load	2025-07-22 21:17:27 -04:00
Wing Lian	5c74bebfd0	use new upstream branches for nd-parallelism	2025-07-22 21:12:22 -04:00