bump hf deps (#2735) [skip ci]

* bump hf deps * upgrade liger-kernel too * install cce from fork for transformers fix * fix reference to vocab size in gemma3 patch * use padding_idx instead of pad_token_id * remove fixed gemma3 patch * use updated cce fork * fix local mllama cce patches w docstring * add test for multipack with trainer setup and fix trainer for trainer refactor upstream * bump modal version * guard for iterable datasetS * mllama model arch layout changed in latest transformers * fix batch sampler with drop_last * fix: address upstream vlm changes for lora * fix: update references to old lora target path * fix: remove mllama fa2 patch due to upstream fix * fix: lora kernel patch path for multimodal models * fix: removed mllama from quarto * run test for came optim on 2.6.0+ * fix fsdp2 patch and remove deprecated patch * make sure to set sequence_parallel_degree for grpo * Add SP test for GRPO * add sp to grpo config for trainer * use reward_funcs as kwarg to grpo trainer * fix the comprehension for reward funcs * reward funcs already passed in as args * init sp_group right before training * fix check for adding models to SP context * make sure to pass args to super * upgrade deepspeed * use updated trl and add reasoning flags for vllm * patch the worker --------- Co-authored-by: NanoCode012 <nano@axolotl.ai>
2025-06-05 07:20:33 -07:00
parent 787880215b
commit c67910fa6f
33 changed files with 470 additions and 695 deletions
--- a/tests/e2e/integrations/test_kd.py
+++ b/tests/e2e/integrations/test_kd.py
@@ -90,7 +90,7 @@ class TestKnowledgeDistillation:
        train(cfg=cfg, dataset_meta=dataset_meta)
        assert (Path(temp_dir) / "model.safetensors").exists()
        check_tensorboard(
-            temp_dir + "/runs", "train/loss", 1.2, "Train Loss (%s) is too high"
+            temp_dir + "/runs", "train/loss", 1.4, "Train Loss (%s) is too high"
        )

    @pytest.mark.parametrize(
--- a/tests/e2e/multigpu/solo/test_grpo.py
+++ b/tests/e2e/multigpu/solo/test_grpo.py
@@ -262,6 +262,99 @@ def oai_gsm8k_transform(cfg, *args, **kwargs):
                    **current_env,
                },
            )
+        finally:
+            (recursive_kill(vllm_process))
+
+    @require_vllm
+    def test_llama_lora_sp(self, temp_dir):
+        rnd_reward_suffix = str(random.randint(1000, 9999))
+        cfg = DictDefault(
+            {
+                "base_model": "HuggingFaceTB/SmolLM2-135M",
+                "chat_template": "llama3",
+                "rl": "grpo",
+                "trl": {
+                    "beta": 0.001,
+                    "max_completion_length": 256,
+                    "use_vllm": True,
+                    "num_generations": 4,
+                    "reward_funcs": [f"rewards_{rnd_reward_suffix}.rand_reward_func"],
+                },
+                "vllm": {
+                    "max_model_len": 800,
+                    "enable_prefix_caching": True,
+                },
+                "datasets": [
+                    {
+                        "path": "openai/gsm8k",
+                        "name": "main",
+                        "type": f"rewards_{rnd_reward_suffix}.oai_gsm8k_transform",
+                    },
+                ],
+                "adapter": "lora",
+                "lora_r": 8,
+                "lora_alpha": 16,
+                "lora_dropout": 0.05,
+                "lora_target_linear": True,
+                "sequence_parallel_degree": 2,
+                "flash_attention": True,
+                "sequence_len": 1024,
+                "special_tokens": {
+                    "pad_token": "<|endoftext|>",
+                },
+                "max_steps": 3,
+                "num_epochs": 1,
+                "micro_batch_size": 4,
+                "gradient_accumulation_steps": 2,
+                "warmup_steps": 10,
+                "val_set_size": 0.0,
+                "output_dir": temp_dir,
+                "learning_rate": 0.0001,
+                "optimizer": "adamw_torch_fused",
+                "lr_scheduler": "cosine",
+                "save_safetensors": True,
+                "bf16": "auto",
+                "use_tensorboard": True,
+            }
+        )
+
+        self._utils_write_yaml_and_rewards(cfg, temp_dir, suffix=rnd_reward_suffix)
+
+        current_env = os.environ.copy()
+        env = {
+            "NCCL_P2P_LEVEL": "LOC",
+            **current_env,
+            "CUDA_VISIBLE_DEVICES": "1",
+        }
+        vllm_process = start_vllm(
+            cfg.base_model,
+            env=env,
+            quiet=True,
+            wait=300,
+            gpu_memory_utilization=0.15,
+            max_model_len=cfg.vllm.max_model_len,
+            enable_prefix_caching=cfg.vllm.enable_prefix_caching,
+            host="0.0.0.0",
+            port=8000,
+        )
+
+        try:
+            execute_subprocess_async(
+                [
+                    "axolotl",
+                    "train",
+                    str(Path(temp_dir) / "config.yaml"),
+                    "--num-processes",
+                    str(2),
+                    "--main-process-port",
+                    f"{get_torch_dist_unique_port()}",
+                ],
+                env={
+                    "NCCL_P2P_LEVEL": "LOC",
+                    "NCCL_DEBUG": "INFO",
+                    **current_env,
+                },
+            )
        finally:
            recursive_kill(vllm_process)

--- a/tests/e2e/test_llama_vision.py
+++ b/tests/e2e/test_llama_vision.py
@@ -33,7 +33,7 @@ class TestLlamaVision(unittest.TestCase):
                "lora_r": 8,
                "lora_alpha": 16,
                "lora_dropout": 0.05,
-                "lora_target_modules": r"language_model.model.layers.[\d]+.(mlp|cross_attn|self_attn).(up|down|gate|q|k|v|o)_proj",
+                "lora_target_modules": r"model.language_model.layers.[\d]+.(mlp|cross_attn|self_attn).(up|down|gate|q|k|v|o)_proj",
                "val_set_size": 0,
                "chat_template": "llama3_2_vision",
                "datasets": [
@@ -81,7 +81,7 @@ class TestLlamaVision(unittest.TestCase):
                "lora_r": 8,
                "lora_alpha": 16,
                "lora_dropout": 0.05,
-                "lora_target_modules": r"language_model.model.layers.[\d]+.(mlp|cross_attn|self_attn).(up|down|gate|q|k|v|o)_proj",
+                "lora_target_modules": r"model.language_model.layers.[\d]+.(mlp|cross_attn|self_attn).(up|down|gate|q|k|v|o)_proj",
                "val_set_size": 0,
                "chat_template": "llama3_2_vision",
                "datasets": [
--- a/tests/e2e/test_optimizers.py
+++ b/tests/e2e/test_optimizers.py
@@ -10,7 +10,12 @@ from axolotl.train import train
 from axolotl.utils.config import normalize_config, validate_config
 from axolotl.utils.dict import DictDefault

-from .utils import check_model_output_exists, require_torch_2_5_1, with_temp_dir
+from .utils import (
+    check_model_output_exists,
+    require_torch_2_5_1,
+    require_torch_2_6_0,
+    with_temp_dir,
+)


 class TestCustomOptimizers(unittest.TestCase):
@@ -196,6 +201,7 @@ class TestCustomOptimizers(unittest.TestCase):
        check_model_output_exists(temp_dir, cfg)

    @with_temp_dir
+    @require_torch_2_6_0
    def test_came_pytorch(self, temp_dir):
        # pylint: disable=duplicate-code
        cfg = DictDefault(
--- a/tests/test_packed_dataset.py
+++ b/tests/test_packed_dataset.py
@@ -6,10 +6,16 @@ from pathlib import Path
 from datasets import Dataset, load_dataset
 from transformers import AutoTokenizer

+from axolotl.cli.args import TrainerCliArgs
+from axolotl.common.datasets import load_datasets
 from axolotl.datasets import ConstantLengthDataset, TokenizedPromptDataset
 from axolotl.prompt_tokenizers import AlpacaPromptTokenizingStrategy
 from axolotl.prompters import AlpacaPrompter
+from axolotl.train import setup_model_and_trainer
+from axolotl.utils.config import normalize_config, validate_config
+from axolotl.utils.dict import DictDefault

+from tests.e2e.utils import with_temp_dir
 from tests.hf_offline_utils import enable_hf_offline


@@ -67,6 +73,85 @@ class TestPacking(unittest.TestCase):
        assert example["position_ids"][next_bos_index] == 0
        assert example["position_ids"][next_bos_index + 1] == 1

+    @with_temp_dir
+    def test_lora_packing(self, temp_dir):
+        # pylint: disable=duplicate-code
+        cfg = DictDefault(
+            {
+                "base_model": "HuggingFaceTB/SmolLM2-135M",
+                "tokenizer_type": "AutoTokenizer",
+                "sequence_len": 1024,
+                "sample_packing": True,
+                "multipack_real_batches": False,
+                "eval_sample_packing": True,
+                "adapter": "lora",
+                "lora_r": 32,
+                "lora_alpha": 64,
+                "lora_dropout": 0.05,
+                "lora_target_linear": True,
+                "val_set_size": 0.2,
+                "special_tokens": {
+                    "pad_token": "<|endoftext|>",
+                },
+                "datasets": [
+                    {
+                        "path": "mhenrichsen/alpaca_2k_test",
+                        "type": "alpaca",
+                    },
+                ],
+                "num_epochs": 1,
+                "max_steps": 20,
+                "save_steps": 10,
+                "micro_batch_size": 8,
+                "gradient_accumulation_steps": 1,
+                "output_dir": temp_dir,
+                "learning_rate": 0.00001,
+                "optimizer": "adamw_torch_fused",
+                "lr_scheduler": "cosine",
+                "fp16": False,
+                "bf16": False,
+            }
+        )
+
+        cfg = validate_config(cfg)
+        normalize_config(cfg)
+        cli_args = TrainerCliArgs()
+        dataset_meta = load_datasets(cfg=cfg, cli_args=cli_args)
+
+        (
+            trainer,
+            _,
+            _,
+            _,
+            _,
+        ) = setup_model_and_trainer(cfg, dataset_meta)
+
+        sampler = trainer._get_eval_sampler(  # pylint: disable=protected-access
+            trainer.eval_dataset
+        )
+        assert "MultipackBatchSampler" in sampler.__class__.__name__
+        assert (
+            "V2BatchSamplerDataCollatorForSeq2Seq"
+            in trainer.eval_data_collator.__class__.__name__
+        )
+        dataloader = trainer.get_eval_dataloader(trainer.eval_dataset)
+        dataloader_iter = iter(dataloader)
+        batch = next(dataloader_iter)
+        assert batch["input_ids"].shape == (1, 8192)
+
+        sampler = trainer._get_train_sampler(  # pylint: disable=protected-access
+            trainer.train_dataset
+        )
+        assert "MultipackBatchSampler" in sampler.__class__.__name__
+        assert (
+            "V2BatchSamplerDataCollatorForSeq2Seq"
+            in trainer.train_data_collator.__class__.__name__
+        )
+        dataloader = trainer.get_train_dataloader()
+        dataloader_iter = iter(dataloader)
+        batch = next(dataloader_iter)
+        assert batch["input_ids"].shape == (1, 8192)
+

 if __name__ == "__main__":
    unittest.main()