pre-patch the mlp

use new patch
wip patch
2025-07-13 23:01:49 -04:00 · 2025-07-13 22:40:37 -04:00 · 2025-07-13 22:37:18 -04:00 · 2025-07-13 22:37:18 -04:00
211 changed files with 798 additions and 2353 deletions
--- a/.coderabbit.yaml
+++ b/.coderabbit.yaml
@@ -1,16 +0,0 @@
-# yaml-language-server: $schema=https://coderabbit.ai/integrations/schema.v2.json
-language: "en-US"
-early_access: false
-reviews:
-  profile: "chill"
-  request_changes_workflow: false
-  high_level_summary: true
-  review_status: true
-  collapse_walkthrough: true
-  poem: false
-  sequence_diagrams: false
-  auto_review:
-    enabled: true
-    drafts: false
-chat:
-  auto_reply: true
--- a/.github/workflows/main.yml
+++ b/.github/workflows/main.yml
@@ -87,6 +87,7 @@ jobs:
            python_version: "3.11"
            pytorch: 2.6.0
            axolotl_extras:
+            is_latest: true
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
@@ -97,7 +98,6 @@ jobs:
            python_version: "3.11"
            pytorch: 2.7.1
            axolotl_extras:
-            is_latest: true
          - cuda: 128
            cuda_version: 12.8.1
            python_version: "3.11"
--- a/.github/workflows/multi-gpu-e2e.yml
+++ b/.github/workflows/multi-gpu-e2e.yml
@@ -33,13 +33,6 @@ jobs:
            axolotl_extras:
            num_gpus: 2
            nightly_build: "true"
-          - cuda: 126
-            cuda_version: 12.6.3
-            python_version: "3.11"
-            pytorch: 2.7.0
-            axolotl_extras: vllm
-            num_gpus: 2
-            nightly_build: "true"
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
--- a/.github/workflows/nightlies.yml
+++ b/.github/workflows/nightlies.yml
@@ -12,16 +12,11 @@ jobs:
      fail-fast: false
      matrix:
        include:
-          - cuda: 126
-            cuda_version: 12.6.3
+          - cuda: 124
+            cuda_version: 12.4.1
            python_version: "3.11"
            pytorch: 2.6.0
            axolotl_extras:
-          - cuda: 126
-            cuda_version: 12.6.3
-            python_version: "3.11"
-            pytorch: 2.7.1
-            axolotl_extras:
    runs-on: axolotl-gpu-runner
    steps:
      - name: Checkout
@@ -65,15 +60,15 @@ jobs:
    strategy:
      matrix:
        include:
-          - cuda: 126
-            cuda_version: 12.6.3
+          - cuda: 124
+            cuda_version: 12.4.1
            python_version: "3.11"
            pytorch: 2.6.0
            axolotl_extras:
          - cuda: 126
            cuda_version: 12.6.3
            python_version: "3.11"
-            pytorch: 2.7.1
+            pytorch: 2.6.0
            axolotl_extras:
    runs-on: axolotl-gpu-runner
    steps:
--- a/.github/workflows/tests-nightly.yml
+++ b/.github/workflows/tests-nightly.yml
@@ -92,7 +92,7 @@ jobs:
    if: github.repository_owner == 'axolotl-ai-cloud'
    # this job needs to be run on self-hosted GPU runners...
    runs-on: [self-hosted, modal]
-    timeout-minutes: 120
+    timeout-minutes: 60
    needs: [pre-commit, pytest]

    strategy:
@@ -106,13 +106,6 @@ jobs:
            num_gpus: 1
            axolotl_extras:
            nightly_build: "true"
-          - cuda: 126
-            cuda_version: 12.6.3
-            python_version: "3.11"
-            pytorch: 2.7.1
-            num_gpus: 1
-            axolotl_extras:
-            nightly_build: "true"
    steps:
      - name: Checkout
        uses: actions/checkout@v4
@@ -123,7 +116,7 @@ jobs:
      - name: Install Modal
        run: |
          python -m pip install --upgrade pip
-          pip install modal==1.0.2 jinja2
+          pip install modal==0.71.8 jinja2
      - name: Update env vars
        run: |
          echo "BASE_TAG=main-base-py${{ matrix.python_version }}-cu${{ matrix.cuda }}-${{ matrix.pytorch }}" >> $GITHUB_ENV
@@ -137,45 +130,3 @@ jobs:
      - name: Run tests job on Modal
        run: |
          modal run cicd.e2e_tests
-  docker-e2e-multigpu-tests:
-    if: github.repository_owner == 'axolotl-ai-cloud'
-    # this job needs to be run on self-hosted GPU runners...
-    runs-on: [self-hosted, modal]
-    timeout-minutes: 120
-    needs: [pre-commit, pytest, docker-e2e-tests]
-
-    strategy:
-      fail-fast: false
-      matrix:
-        include:
-          - cuda: 126
-            cuda_version: 12.6.3
-            python_version: "3.11"
-            pytorch: 2.7.1
-            num_gpus: 2
-            axolotl_extras:
-            nightly_build: "true"
-    steps:
-      - name: Checkout
-        uses: actions/checkout@v4
-      - name: Install Python
-        uses: actions/setup-python@v5
-        with:
-          python-version: "3.11"
-      - name: Install Modal
-        run: |
-          python -m pip install --upgrade pip
-          pip install modal==1.0.2 jinja2
-      - name: Update env vars
-        run: |
-          echo "BASE_TAG=main-base-py${{ matrix.python_version }}-cu${{ matrix.cuda }}-${{ matrix.pytorch }}" >> $GITHUB_ENV
-          echo "PYTORCH_VERSION=${{ matrix.pytorch}}" >> $GITHUB_ENV
-          echo "AXOLOTL_ARGS=${{ matrix.axolotl_args}}" >> $GITHUB_ENV
-          echo "AXOLOTL_EXTRAS=${{ matrix.axolotl_extras}}" >> $GITHUB_ENV
-          echo "CUDA=${{ matrix.cuda }}" >> $GITHUB_ENV
-          echo "N_GPUS=${{ matrix.num_gpus }}" >> $GITHUB_ENV
-          echo "NIGHTLY_BUILD=${{ matrix.nightly_build }}" >> $GITHUB_ENV
-          echo "CODECOV_TOKEN=${{ secrets.CODECOV_TOKEN }}" >> $GITHUB_ENV
-      - name: Run tests job on Modal
-        run: |
-          modal run cicd.multigpu
--- a/_quarto.yml
+++ b/_quarto.yml
@@ -276,7 +276,6 @@ website:
            - docs/torchao.qmd
            - docs/custom_integrations.qmd
            - docs/sequence_parallelism.qmd
-            - docs/gradient_checkpointing.qmd

        - section: "Troubleshooting"
          contents:
--- a/codecov.yml
+++ b/codecov.yml
@@ -22,7 +22,6 @@ coverage:
        only_pulls: true
        flags: null
        paths: null
-        informational: true
    patch:
      default:
        # basic
--- a/docs/gradient_checkpointing.qmd
+++ b/docs/gradient_checkpointing.qmd
@@ -1,29 +0,0 @@
---
-title: Gradient Checkpointing and Activation Offloading
---
-
-Gradient checkpointing and activation offloading are techniques used to optimize the performance of deep learning
-models by reducing the memory footprint and improving computational efficiency.
-
-### Enabling Gradient Checkpointing
-
-```yaml
-gradient_checkpointing: true
-```
-
-### Enabling Activation Offloading
-
-```yaml
-gradient_checkpointing: true  # required for activation offloading
-activation_offloading: true
-```
-
-Activation offloading variants:
-
-The default `activation_offloading: true` offloads activations to CPU and uses CUDA streams
-to overlap the communications and computations when offloading.
-
-The `activation_offloading: legacy` naively offloads activations to CPU and without additional optimizations.
-
-For resource constrained environments with limited CPU memory, `activation_offloading: disk` offloads
-activations to disk instead of CPU RAM so that much larger context lengths can be trained with minimal memory.
--- a/examples/cloud/modal.yaml
+++ b/examples/cloud/modal.yaml
@@ -26,5 +26,3 @@ timeout: 86400
 # Preprocess specific configurations
 memory_preprocess: 32
 timeout_preprocess: 14400
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/cohere/command-r-7b-qlora.yml
+++ b/examples/cohere/command-r-7b-qlora.yml
@@ -35,6 +35,7 @@ wandb_watch:
 wandb_name:
 wandb_log_model:

+
 gradient_accumulation_steps: 4
 micro_batch_size: 1
 num_epochs: 4
@@ -55,5 +56,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/colab-notebooks/colab-axolotl-example.ipynb
+++ b/examples/colab-notebooks/colab-axolotl-example.ipynb
@@ -40,7 +40,7 @@
        "%%capture\n",
        "# This step can take ~5-10 minutes to install dependencies\n",
        "!pip install --no-build-isolation axolotl[flash-attn]>=0.9.1\n",
-        "!pip install \"cut-cross-entropy[transformers] @ git+https://github.com/axolotl-ai-cloud/ml-cross-entropy.git@50cef19\""
+        "!pip install \"cut-cross-entropy[transformers] @ git+https://github.com/axolotl-ai-cloud/ml-cross-entropy.git@78b2a45713a54c9bedf8b33f5e31cf07a1a57154\""
      ]
    },
    {
--- a/examples/deepcogito/cogito-v1-preview-llama-3B-lora.yml
+++ b/examples/deepcogito/cogito-v1-preview-llama-3B-lora.yml
@@ -56,5 +56,3 @@ evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/deepcogito/cogito-v1-preview-qwen-14B-lora.yml
+++ b/examples/deepcogito/cogito-v1-preview-qwen-14B-lora.yml
@@ -56,5 +56,3 @@ evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/deepseek-v2/fft-fsdp-16b.yaml
+++ b/examples/deepseek-v2/fft-fsdp-16b.yaml
@@ -55,5 +55,3 @@ fsdp_config:
  fsdp_transformer_layer_cls_to_wrap: DeepseekV2DecoderLayer
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_sharding_strategy: FULL_SHARD
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/deepseek-v2/qlora-fsdp-2_5.yaml
+++ b/examples/deepseek-v2/qlora-fsdp-2_5.yaml
@@ -79,5 +79,3 @@ fsdp_config:
  fsdp_transformer_layer_cls_to_wrap: DeepseekV2DecoderLayer
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_sharding_strategy: FULL_SHARD
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/devstral/devstral-small-qlora.yml
+++ b/examples/devstral/devstral-small-qlora.yml
@@ -62,5 +62,3 @@ saves_per_epoch: 1

 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/falcon-h1/falcon-h1-1b-deep-qlora.yaml
+++ b/examples/falcon-h1/falcon-h1-1b-deep-qlora.yaml
@@ -69,5 +69,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/falcon-h1/falcon-h1-1b-qlora.yaml
+++ b/examples/falcon-h1/falcon-h1-1b-qlora.yaml
@@ -46,6 +46,7 @@ wandb_watch:
 wandb_name:
 wandb_log_model:

+
 gradient_accumulation_steps: 4
 micro_batch_size: 1
 num_epochs: 4
@@ -68,5 +69,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/falcon-h1/falcon-h1-34b-qlora.yaml
+++ b/examples/falcon-h1/falcon-h1-34b-qlora.yaml
@@ -69,5 +69,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/falcon-h1/falcon-h1-3b-qlora.yaml
+++ b/examples/falcon-h1/falcon-h1-3b-qlora.yaml
@@ -69,5 +69,3 @@ evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/falcon-h1/falcon-h1-500m-qlora.yaml
+++ b/examples/falcon-h1/falcon-h1-500m-qlora.yaml
@@ -69,5 +69,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/falcon-h1/falcon-h1-7b-qlora.yaml
+++ b/examples/falcon-h1/falcon-h1-7b-qlora.yaml
@@ -69,5 +69,3 @@ evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/gemma2/qlora.yml
+++ b/examples/gemma2/qlora.yml
@@ -60,5 +60,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/gemma2/reward-model.yaml
+++ b/examples/gemma2/reward-model.yaml
@@ -50,5 +50,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/gemma3/gemma-3-1b-qlora.yml
+++ b/examples/gemma3/gemma-3-1b-qlora.yml
@@ -66,5 +66,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/gemma3/gemma-3-4b-qlora.yml
+++ b/examples/gemma3/gemma-3-4b-qlora.yml
@@ -60,5 +60,3 @@ warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/gemma3/gemma-3-4b-vision-qlora.yml
+++ b/examples/gemma3/gemma-3-4b-vision-qlora.yml
@@ -62,5 +62,3 @@ warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/glm4/qlora-32b.yaml
+++ b/examples/glm4/qlora-32b.yaml
@@ -60,5 +60,3 @@ evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/jamba/qlora.yaml
+++ b/examples/jamba/qlora.yaml
@@ -54,5 +54,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/jamba/qlora_deepspeed.yaml
+++ b/examples/jamba/qlora_deepspeed.yaml
@@ -55,5 +55,3 @@ saves_per_epoch: 1
 deepspeed: deepspeed_configs/zero2.json
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/jamba/qlora_fsdp_large.yaml
+++ b/examples/jamba/qlora_fsdp_large.yaml
@@ -64,5 +64,3 @@ fsdp_config:
  fsdp_transformer_layer_cls_to_wrap: JambaAttentionDecoderLayer,JambaMambaDecoderLayer
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_sharding_strategy: FULL_SHARD
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/lfm2/lfm2-350m-fft.yaml
+++ b/examples/lfm2/lfm2-350m-fft.yaml
@@ -46,5 +46,3 @@ evals_per_epoch: 2
 saves_per_epoch: 1

 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/fft_optimized.yml
+++ b/examples/llama-2/fft_optimized.yml
@@ -55,5 +55,3 @@ saves_per_epoch: 1
 deepspeed: #deepspeed_configs/zero2.json # multi-gpu only
 weight_decay: 0.1
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/gptq-lora.yml
+++ b/examples/llama-2/gptq-lora.yml
@@ -64,5 +64,3 @@ special_tokens:
  bos_token: "<s>"
  eos_token: "</s>"
  unk_token: "<unk>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/lisa.yml
+++ b/examples/llama-2/lisa.yml
@@ -60,5 +60,3 @@ special_tokens:
  bos_token: "<s>"
  eos_token: "</s>"
  unk_token: "<unk>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/loftq.yml
+++ b/examples/llama-2/loftq.yml
@@ -52,5 +52,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/lora.yml
+++ b/examples/llama-2/lora.yml
@@ -52,5 +52,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/qlora-fsdp.yml
+++ b/examples/llama-2/qlora-fsdp.yml
@@ -67,5 +67,3 @@ fsdp_config:
  fsdp_transformer_layer_cls_to_wrap: LlamaDecoderLayer
  fsdp_state_dict_type: FULL_STATE_DICT
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/qlora.yml
+++ b/examples/llama-2/qlora.yml
@@ -53,5 +53,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-2/relora.yml
+++ b/examples/llama-2/relora.yml
@@ -58,5 +58,3 @@ special_tokens:
  bos_token: "<s>"
  eos_token: "</s>"
  unk_token: "<unk>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3-vision/lora-11b.yaml
+++ b/examples/llama-3-vision/lora-11b.yaml
@@ -57,5 +57,3 @@ warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/3b-qat-fsdp2.yaml
+++ b/examples/llama-3/3b-qat-fsdp2.yaml
@@ -77,5 +77,3 @@ fsdp_config:

 special_tokens:
  pad_token: <|end_of_text|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/fft-8b-liger-fsdp.yaml
+++ b/examples/llama-3/fft-8b-liger-fsdp.yaml
@@ -72,5 +72,3 @@ fsdp_config:
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot_id|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/fft-8b.yaml
+++ b/examples/llama-3/fft-8b.yaml
@@ -42,5 +42,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: <|end_of_text|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/instruct-dpo-lora-8b.yml
+++ b/examples/llama-3/instruct-dpo-lora-8b.yml
@@ -71,5 +71,3 @@ warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/instruct-lora-8b.yml
+++ b/examples/llama-3/instruct-lora-8b.yml
@@ -64,5 +64,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
   pad_token: <|end_of_text|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/lora-1b-deduplicate-dpo.yml
+++ b/examples/llama-3/lora-1b-deduplicate-dpo.yml
@@ -83,5 +83,3 @@ warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/lora-1b-deduplicate-sft.yml
+++ b/examples/llama-3/lora-1b-deduplicate-sft.yml
@@ -61,5 +61,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
   pad_token: <|end_of_text|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/lora-1b-kernels.yml
+++ b/examples/llama-3/lora-1b-kernels.yml
@@ -65,5 +65,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: "<|end_of_text|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/lora-1b-ray.yml
+++ b/examples/llama-3/lora-1b-ray.yml
@@ -64,5 +64,3 @@ special_tokens:

 use_ray: true
 ray_num_workers: 4
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/lora-1b-sample-packing-sequentially.yml
+++ b/examples/llama-3/lora-1b-sample-packing-sequentially.yml
@@ -63,5 +63,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: <|end_of_text|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/lora-1b.yml
+++ b/examples/llama-3/lora-1b.yml
@@ -60,5 +60,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: "<|end_of_text|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/lora-8b.yml
+++ b/examples/llama-3/lora-8b.yml
@@ -57,5 +57,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
   pad_token: <|end_of_text|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/qlora-1b-kto.yaml
+++ b/examples/llama-3/qlora-1b-kto.yaml
@@ -61,5 +61,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: "<|end_of_text|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/qlora-1b.yml
+++ b/examples/llama-3/qlora-1b.yml
@@ -62,5 +62,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: "<|end_of_text|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/qlora-fsdp-405b.yaml
+++ b/examples/llama-3/qlora-fsdp-405b.yaml
@@ -60,5 +60,3 @@ fsdp_config:
  fsdp_sharding_strategy: FULL_SHARD
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/qlora-fsdp-70b.yaml
+++ b/examples/llama-3/qlora-fsdp-70b.yaml
@@ -69,5 +69,3 @@ fsdp_config:
  fsdp_sharding_strategy: FULL_SHARD
 special_tokens:
  pad_token: <|end_of_text|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/qlora.yml
+++ b/examples/llama-3/qlora.yml
@@ -54,5 +54,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: "<|end_of_text|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-3/sparse-finetuning.yaml
+++ b/examples/llama-3/sparse-finetuning.yaml
@@ -75,5 +75,3 @@ llmcompressor:
          ]
          start: 0
  save_compressed: true
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-4/do-no-use-fa2/maverick-qlora-fsdp1.yaml
+++ b/examples/llama-4/do-no-use-fa2/maverick-qlora-fsdp1.yaml
@@ -86,5 +86,3 @@ fsdp_config:
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-4/do-no-use-fa2/scout-qlora-fsdp1.yaml
+++ b/examples/llama-4/do-no-use-fa2/scout-qlora-fsdp1.yaml
@@ -90,5 +90,3 @@ fsdp_config:
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-4/do-no-use-fa2/scout-qlora-single-h100.yaml
+++ b/examples/llama-4/do-no-use-fa2/scout-qlora-single-h100.yaml
@@ -83,5 +83,3 @@ weight_decay: 0.0
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-4/do-no-use-fa2/scout-vision-qlora-fsdp.yaml
+++ b/examples/llama-4/do-no-use-fa2/scout-vision-qlora-fsdp.yaml
@@ -86,5 +86,3 @@ fsdp_config:
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-4/scout-qlora-flexattn-fsdp2.yaml
+++ b/examples/llama-4/scout-qlora-flexattn-fsdp2.yaml
@@ -84,5 +84,3 @@ fsdp_config:
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-4/scout-qlora-single-h100-flex.yaml
+++ b/examples/llama-4/scout-qlora-single-h100-flex.yaml
@@ -82,5 +82,3 @@ weight_decay: 0.0
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llama-4/scout-vision-qlora-fsdp2-flex.yaml
+++ b/examples/llama-4/scout-vision-qlora-fsdp2-flex.yaml
@@ -87,5 +87,3 @@ fsdp_config:
 special_tokens:
  pad_token: <|finetune_right_pad_id|>
  eos_token: <|eot|>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/llava/lora-7b.yaml
+++ b/examples/llava/lora-7b.yaml
@@ -53,5 +53,3 @@ warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/magistral/magistral-small-fsdp-qlora.yaml
+++ b/examples/magistral/magistral-small-fsdp-qlora.yaml
@@ -70,5 +70,3 @@ fsdp_config:
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_transformer_layer_cls_to_wrap: MistralDecoderLayer
  fsdp_activation_checkpointing: true
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/magistral/magistral-small-qlora.yaml
+++ b/examples/magistral/magistral-small-qlora.yaml
@@ -61,5 +61,3 @@ flash_attention: true
 warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mamba/config.yml
+++ b/examples/mamba/config.yml
@@ -48,5 +48,3 @@ weight_decay: 0.0
 special_tokens:
 tokens:
 save_safetensors: False
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/bigstral-ds-zero3.yaml
+++ b/examples/mistral/bigstral-ds-zero3.yaml
@@ -53,5 +53,3 @@ special_tokens:
  eos_token: "<|im_end|>"
 tokens:
  - "<|im_start|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/config.yml
+++ b/examples/mistral/config.yml
@@ -43,5 +43,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/lora-mps.yml
+++ b/examples/mistral/lora-mps.yml
@@ -64,5 +64,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/lora.yml
+++ b/examples/mistral/lora.yml
@@ -64,5 +64,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mistral-dpo-qlora.yml
+++ b/examples/mistral/mistral-dpo-qlora.yml
@@ -80,5 +80,3 @@ weight_decay: 0.0
 special_tokens:
  bos_token: "<|im_start|>"
  eos_token: "<|im_end|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mistral-qlora-fsdp.yml
+++ b/examples/mistral/mistral-qlora-fsdp.yml
@@ -74,5 +74,3 @@ fsdp_config:
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_auto_wrap_policy: TRANSFORMER_BASED_WRAP
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mistral-qlora-orpo.yml
+++ b/examples/mistral/mistral-qlora-orpo.yml
@@ -69,5 +69,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mistral-small-3.1-24B-lora.yml
+++ b/examples/mistral/mistral-small-3.1-24B-lora.yml
@@ -56,5 +56,3 @@ evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mixtral-8x22b-qlora-fsdp.yml
+++ b/examples/mistral/mixtral-8x22b-qlora-fsdp.yml
@@ -72,5 +72,3 @@ fsdp_config:
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_auto_wrap_policy: TRANSFORMER_BASED_WRAP
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mixtral-qlora-fsdp.yml
+++ b/examples/mistral/mixtral-qlora-fsdp.yml
@@ -77,5 +77,3 @@ fsdp_config:
  fsdp_forward_prefetch: false
  fsdp_backward_prefetch: BACKWARD_PRE
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mixtral.yml
+++ b/examples/mistral/mixtral.yml
@@ -81,5 +81,3 @@ saves_per_epoch: 1
 deepspeed: deepspeed_configs/zero2.json
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/mixtral_22.yml
+++ b/examples/mistral/mixtral_22.yml
@@ -51,5 +51,3 @@ special_tokens:
  eos_token: "<|im_end|>"
 tokens:
  - "<|im_start|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/mistral/qlora.yml
+++ b/examples/mistral/qlora.yml
@@ -64,5 +64,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/orpheus/finetune.yml
+++ b/examples/orpheus/finetune.yml
@@ -50,5 +50,3 @@ weight_decay: 0.05

 special_tokens:
  pad_token: <custom_token_7>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/phi/lora-3.5.yaml
+++ b/examples/phi/lora-3.5.yaml
@@ -63,5 +63,3 @@ warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 4
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/phi/phi-ft.yml
+++ b/examples/phi/phi-ft.yml
@@ -57,5 +57,3 @@ weight_decay: 0.1
 resize_token_embeddings_to_32x: true
 special_tokens:
  pad_token: "<|endoftext|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/phi/phi-qlora.yml
+++ b/examples/phi/phi-qlora.yml
@@ -60,5 +60,3 @@ weight_decay: 0.1
 resize_token_embeddings_to_32x: true
 special_tokens:
  pad_token: "<|endoftext|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/phi/phi2-ft.yml
+++ b/examples/phi/phi2-ft.yml
@@ -57,5 +57,3 @@ weight_decay: 0.1
 resize_token_embeddings_to_32x: true
 special_tokens:
  pad_token: "<|endoftext|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/phi/phi3-ft-fsdp.yml
+++ b/examples/phi/phi3-ft-fsdp.yml
@@ -71,5 +71,3 @@ fsdp_config:
 resize_token_embeddings_to_32x: true
 special_tokens:
  pad_token: "<|endoftext|>"
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/phi/phi3-ft.yml
+++ b/examples/phi/phi3-ft.yml
@@ -59,5 +59,3 @@ warmup_ratio: 0.2
 debug: true
 weight_decay: 0.1
 resize_token_embeddings_to_32x: true
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/pixtral/lora-12b.yml
+++ b/examples/pixtral/lora-12b.yml
@@ -55,5 +55,3 @@ saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
  pad_token: <pad>
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen2-vl/lora-7b.yaml
+++ b/examples/qwen2-vl/lora-7b.yaml
@@ -53,5 +53,3 @@ warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen2/dpo.yaml
+++ b/examples/qwen2/dpo.yaml
@@ -54,5 +54,3 @@ warmup_steps: 10
 evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen2/prm.yaml
+++ b/examples/qwen2/prm.yaml
@@ -55,5 +55,3 @@ eval_steps: 100
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen2/qlora-fsdp.yaml
+++ b/examples/qwen2/qlora-fsdp.yaml
@@ -67,5 +67,3 @@ fsdp_config:
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_sharding_strategy: FULL_SHARD
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen2/reward-model.yaml
+++ b/examples/qwen2/reward-model.yaml
@@ -26,6 +26,7 @@ wandb_watch:
 wandb_name:
 wandb_log_model:

+
 gradient_accumulation_steps: 4
 micro_batch_size: 2
 num_epochs: 4
@@ -49,5 +50,3 @@ evals_per_epoch:
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen2_5-vl/lora-7b.yaml
+++ b/examples/qwen2_5-vl/lora-7b.yaml
@@ -53,5 +53,3 @@ warmup_ratio: 0.1
 evals_per_epoch: 1
 saves_per_epoch: 1
 weight_decay: 0.0
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen3/32b-qlora.yaml
+++ b/examples/qwen3/32b-qlora.yaml
@@ -67,5 +67,3 @@ evals_per_epoch: 4
 saves_per_epoch: 1
 weight_decay: 0.0
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen3/8b-qat-fsdp2.yml
+++ b/examples/qwen3/8b-qat-fsdp2.yml
@@ -76,5 +76,3 @@ fsdp_config:
  fsdp_activation_checkpointing: true

 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/examples/qwen3/qlora-fsdp.yaml
+++ b/examples/qwen3/qlora-fsdp.yaml
@@ -66,5 +66,3 @@ fsdp_config:
  fsdp_state_dict_type: FULL_STATE_DICT
  fsdp_sharding_strategy: FULL_SHARD
 special_tokens:
-
-# save_first_step: true  # uncomment this to validate checkpoint saving works with your config
--- a/Show More
+++ b/Show More
Author	SHA1	Message	Date
Wing Lian	6978f09760	pre-patch the mlp	2025-07-13 23:01:49 -04:00
Wing Lian	d41b3814d0	use new patch	2025-07-13 22:40:37 -04:00
Wing Lian	1649f91cd4	wip patch	2025-07-13 22:37:18 -04:00
Wing Lian	5a063f5c75	wip state dict compatible fused mlp	2025-07-13 22:37:18 -04:00