Merge branch 'main' into hymba_multipack2

make sure padding is labeled as -100 for pretraining (#2227 )
fix: allow trainer builder to use custom jinja chat template (#2219 )
2025-01-05 23:15:57 -05:00 · 2024-12-31 15:22:18 -05:00 · 2024-12-24 16:18:50 -05:00 · 2024-12-23 09:08:28 -05:00 · 2024-12-23 07:48:41 -05:00 · 2024-12-22 12:11:39 -05:00
18 changed files with 629 additions and 222 deletions
--- a/.gitignore
+++ b/.gitignore
@@ -1,6 +1,7 @@
 **/axolotl.egg-info
 configs
 last_run_prepared/
+outputs
 .vscode
 _site/

--- a/deepspeed_configs/zero1_torch_compile.json
+++ b/deepspeed_configs/zero1_torch_compile.json
@@ -0,0 +1,27 @@
+{
+  "zero_optimization": {
+    "stage": 1,
+    "overlap_comm": true
+  },
+  "bf16": {
+    "enabled": "auto"
+  },
+  "fp16": {
+    "enabled": "auto",
+    "auto_cast": false,
+    "loss_scale": 0,
+    "initial_scale_power": 32,
+    "loss_scale_window": 1000,
+    "hysteresis": 2,
+    "min_loss_scale": 1
+  },
+  "compile": {
+    "disable": false,
+    "backend": "inductor"
+  },
+  "gradient_accumulation_steps": "auto",
+  "gradient_clipping": "auto",
+  "train_batch_size": "auto",
+  "train_micro_batch_size_per_gpu": "auto",
+  "wall_clock_breakdown": false
+}
--- a/examples/hymba/fft-1.5b.yml
+++ b/examples/hymba/fft-1.5b.yml
@@ -0,0 +1,58 @@
+base_model: nvidia/Hymba-1.5B-Base
+
+load_in_8bit: false
+load_in_4bit: false
+strict: false
+
+datasets:
+  - path: tatsu-lab/alpaca
+    type: alpaca
+dataset_prepared_path: last_run_prepared
+val_set_size: 0.05
+output_dir: ./outputs/out
+
+sequence_len: 2048
+sample_packing: true
+pad_to_sequence_len: true
+
+wandb_project:
+wandb_entity:
+wandb_watch:
+wandb_name:
+wandb_log_model:
+
+gradient_accumulation_steps: 2
+micro_batch_size: 2
+num_epochs: 1
+optimizer: paged_adamw_8bit
+lr_scheduler: cosine
+learning_rate: 2e-5
+
+train_on_inputs: false
+group_by_length: false
+bf16: auto
+fp16:
+tf32: false
+
+trust_remote_code: true
+
+gradient_checkpointing: true
+gradient_checkpointing_kwargs:
+  use_reentrant: false
+early_stopping_patience:
+resume_from_checkpoint:
+logging_steps: 1
+xformers_attention:
+flash_attention: true
+
+warmup_steps: 5
+evals_per_epoch: 2
+eval_table_size:
+saves_per_epoch: 1
+debug:
+deepspeed:
+weight_decay: 0.0
+fsdp:
+fsdp_config:
+special_tokens:
+  pad_token: <|end_of_text|>
--- a/examples/hymba/qlora-1.5b.yml
+++ b/examples/hymba/qlora-1.5b.yml
@@ -0,0 +1,73 @@
+base_model: nvidia/Hymba-1.5B-Base
+
+load_in_8bit: false
+load_in_4bit: True
+strict: false
+
+datasets:
+  - path: tatsu-lab/alpaca
+    type: alpaca
+dataset_prepared_path: last_run_prepared
+val_set_size: 0.05
+output_dir: ./outputs/out
+
+sequence_len: 2048
+sample_packing: true
+pad_to_sequence_len: true
+
+adapter: qlora
+lora_r: 32
+lora_alpha: 16
+lora_dropout: 0.05
+lora_target_linear: true
+lora_fan_in_fan_out:
+lora_target_modules:
+  - gate_proj
+  - down_proj
+  - up_proj
+  - q_proj
+  - v_proj
+  - k_proj
+  - o_proj
+
+wandb_project:
+wandb_entity:
+wandb_watch:
+wandb_name:
+wandb_log_model:
+
+gradient_accumulation_steps: 2
+micro_batch_size: 2
+num_epochs: 1
+optimizer: paged_adamw_8bit
+lr_scheduler: cosine
+learning_rate: 2e-5
+
+train_on_inputs: false
+group_by_length: false
+bf16: auto
+fp16:
+tf32: false
+
+trust_remote_code: true
+
+gradient_checkpointing: true
+gradient_checkpointing_kwargs:
+  use_reentrant: false
+early_stopping_patience:
+resume_from_checkpoint:
+logging_steps: 1
+xformers_attention:
+flash_attention: true
+
+warmup_steps: 5
+evals_per_epoch: 2
+eval_table_size:
+saves_per_epoch: 1
+debug:
+deepspeed:
+weight_decay: 0.0
+fsdp:
+fsdp_config:
+special_tokens:
+  pad_token: <|end_of_text|>
--- a/requirements.txt
+++ b/requirements.txt
@@ -61,4 +61,4 @@ antlr4-python3-runtime==4.13.2
 torchao==0.7.0
 schedulefree==1.3.0

-axolotl-contribs-lgpl==0.0.1b2
+axolotl-contribs-lgpl==0.0.2
--- a/src/axolotl/cli/main.py
+++ b/src/axolotl/cli/main.py
@@ -93,7 +93,7 @@ def evaluate(config: str, accelerate: bool, **kwargs):
@click.argument("config", type=click.Path(exists=True, path_type=str))
@click.option(
    "--accelerate/--no-accelerate",
-    default=True,
+    default=False,
    help="Use accelerate launch for multi-GPU inference",
 )
@click.option(
@@ -124,7 +124,7 @@ def inference(
    if lora_model_dir:
        kwargs["lora_model_dir"] = lora_model_dir
    if base_model:
-        kwargs["output_dir"] = base_model
+        kwargs["base_model"] = base_model

    if accelerate:
        base_cmd = ["accelerate", "launch", "-m", "axolotl.cli.inference"]
--- a/src/axolotl/core/trainer_builder.py
+++ b/src/axolotl/core/trainer_builder.py
@@ -56,6 +56,7 @@ from axolotl.monkeypatch.relora import ReLoRACallback, ReLoRAScheduler
 from axolotl.utils import is_comet_available, is_mlflow_available
 from axolotl.utils.callbacks import (
    EvalFirstStepCallback,
+    GCCallback,
    GPUStatsCallback,
    LossWatchDogCallback,
    SaveAxolotlConfigtoWandBCallback,
@@ -67,7 +68,7 @@ from axolotl.utils.callbacks import (
 )
 from axolotl.utils.callbacks.lisa import lisa_callback_factory
 from axolotl.utils.callbacks.profiler import PytorchProfilerCallback
-from axolotl.utils.chat_templates import get_chat_template
+from axolotl.utils.chat_templates import get_chat_template_from_config
 from axolotl.utils.collators import (
    BatchSamplerDataCollatorForSeq2Seq,
    DataCollatorForSeq2Seq,
@@ -1452,6 +1453,8 @@ class HFCausalTrainerBuilder(TrainerBuilderBase):
        if self.cfg.loss_watchdog_threshold is not None:
            callbacks.append(LossWatchDogCallback(self.cfg))

+        if self.cfg.gc_steps:
+            callbacks.append(GCCallback(gc_steps=self.cfg.gc_steps))
        callbacks.append(SaveModelCallback())

        return callbacks
@@ -1831,8 +1834,8 @@ class HFCausalTrainerBuilder(TrainerBuilderBase):
        training_arguments_kwargs["model_type"] = self.cfg.model_config_type
        training_arguments_kwargs["pretraining"] = bool(self.cfg.pretraining_dataset)
        if self.cfg.chat_template:
-            training_arguments_kwargs["chat_template"] = get_chat_template(
-                self.cfg.chat_template,
+            training_arguments_kwargs["chat_template"] = get_chat_template_from_config(
+                cfg=self.cfg,
                tokenizer=self.tokenizer,
            )

--- a/src/axolotl/monkeypatch/multipack.py
+++ b/src/axolotl/monkeypatch/multipack.py
@@ -25,6 +25,7 @@ SUPPORTED_MULTIPACK_MODEL_TYPES = [
    "gemmoe",
    "starcoder2",
    "deepseek_v2",
+    "hymba",
 ]


--- a/src/axolotl/train.py
+++ b/src/axolotl/train.py
@@ -1,5 +1,6 @@
 """Prepare and train a model on a dataset. Can also infer from a model or merge lora"""

+import inspect
 import os
 import signal
 import sys
@@ -126,7 +127,20 @@ def train(
    )

    if cfg.fix_untrained_tokens:
-        fix_untrained_tokens(model, tokenizer, train_dataset)
+        # check if the `token_ids_to_fix` kwarg exists in the fix_untrained_tokens args
+        sig = inspect.signature(fix_untrained_tokens)
+        # if the function has the `token_ids_to_fix` arg, and fix_untrained_tokens is a list
+        if "token_ids_to_fix" in sig.parameters and isinstance(
+            cfg.fix_untrained_tokens, list
+        ):
+            fix_untrained_tokens(
+                model,
+                tokenizer,
+                train_dataset,
+                token_ids_to_fix=cfg.fix_untrained_tokens,
+            )
+        else:
+            fix_untrained_tokens(model, tokenizer, train_dataset)
        if cfg.local_rank == 0:
            model.save_pretrained(
                str(Path(cfg.output_dir)), safe_serialization=safe_serialization
--- a/src/axolotl/utils/callbacks/init.py
+++ b/src/axolotl/utils/callbacks/init.py
@@ -2,6 +2,7 @@

 from __future__ import annotations

+import gc
 import logging
 import math
 import os
@@ -842,3 +843,17 @@ class SaveModelCallback(TrainerCallback):
    ):
        control.should_save = True
        return control
+
+
+class GCCallback(TrainerCallback):
+    """Callback to garbage collect torch cache"""
+
+    def __init__(self, gc_steps=None):
+        self.gc_steps = gc_steps
+
+    def on_step_end(
+        self, args, state, control, **kwargs  # pylint: disable=unused-argument
+    ):
+        if state.global_step % self.gc_steps == 0:
+            torch.cuda.empty_cache()
+            gc.collect()
--- a/src/axolotl/utils/chat_templates.py
+++ b/src/axolotl/utils/chat_templates.py
@@ -31,6 +31,7 @@ _CHAT_TEMPLATES = {
    "qwen_25": "{%- if tools %}\n    {{- '<|im_start|>system\\n' }}\n    {%- if messages[0]['role'] == 'system' %}\n        {{- messages[0]['content'] }}\n    {%- else %}\n        {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}\n    {%- endif %}\n    {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n    {%- for tool in tools %}\n        {{- \"\\n\" }}\n        {{- tool | tojson }}\n    {%- endfor %}\n    {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n    {%- if messages[0]['role'] == 'system' %}\n        {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n    {%- else %}\n        {{- '<|im_start|>system\\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\\n' }}\n    {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n    {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n        {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n    {%- elif message.role == \"assistant\" %}\n        {{- '<|im_start|>' + message.role }}\n        {%- if message.content %}\n            {{- '\\n' + message.content }}\n        {%- endif %}\n        {%- for tool_call in message.tool_calls %}\n            {%- if tool_call.function is defined %}\n                {%- set tool_call = tool_call.function %}\n            {%- endif %}\n            {{- '\\n<tool_call>\\n{\"name\": \"' }}\n            {{- tool_call.name }}\n            {{- '\", \"arguments\": ' }}\n            {{- tool_call.arguments | tojson }}\n            {{- '}\\n</tool_call>' }}\n        {%- endfor %}\n        {{- '<|im_end|>\\n' }}\n    {%- elif message.role == \"tool\" %}\n        {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n            {{- '<|im_start|>user' }}\n        {%- endif %}\n        {{- '\\n<tool_response>\\n' }}\n        {{- message.content }}\n        {{- '\\n</tool_response>' }}\n        {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n            {{- '<|im_end|>\\n' }}\n        {%- endif %}\n    {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n    {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n",
    "exaone": "{% for message in messages %}{% if loop.first and message['role'] != 'system' %}{{ '[|system|][|endofturn|]\n' }}{% endif %}{{ '[|' + message['role'] + '|]' + message['content'] }}{% if message['role'] == 'user' %}{{ '\n' }}{% else %}{{ '[|endofturn|]\n' }}{% endif %}{% endfor %}{% if add_generation_prompt %}{{ '[|assistant|]' }}{% endif %}",
    "metharme": "{{ bos_token }}{% if messages[0]['role'] == 'system' %}{% set loop_messages = messages[1:] %}{% set system_message = messages[0]['content'] %}{% else %}{% set loop_messages = messages %}{% set system_message = 'Enter RP mode. You shall reply to the user while staying in character. Your responses must be detailed, creative, immersive, and drive the scenario forward.' %}{% endif %}{{ '<|system|>' + system_message }}{% for message in loop_messages %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{{ '<|user|>' + content.strip() }}{% elif message['role'] == 'assistant' %}{{ '<|model|>'  + content.strip() }}{% endif %}{% endfor %}{% if add_generation_prompt %}{{ '<|model|>' }}{% else %}{{ eos_token }}{% endif %}",
+    "hymba": "{{'<extra_id_0>System'}}{% for message in messages %}{% if message['role'] == 'system' %}{{'\n' + message['content'].strip()}}{% if tools or contexts %}{{'\n'}}{% endif %}{% endif %}{% endfor %}{% if tools %}{% for tool in tools %}{{ '\n<tool> ' + tool|tojson + ' </tool>' }}{% endfor %}{% endif %}{% if contexts %}{% if tools %}{{'\n'}}{% endif %}{% for context in contexts %}{{ '\n<context> ' + context.strip() + ' </context>' }}{% endfor %}{% endif %}{{'\n\n'}}{% for message in messages %}{% if message['role'] == 'user' %}{{ '<extra_id_1>User\n' + message['content'].strip() + '\n' }}{% elif message['role'] == 'assistant' %}{{ '<extra_id_1>Assistant\n' + message['content'].strip() + '\n' }}{% elif message['role'] == 'tool' %}{{ '<extra_id_1>Tool\n' + message['content'].strip() + '\n' }}{% endif %}{% endfor %}{%- if add_generation_prompt %}{{'<extra_id_1>Assistant\n'}}{%- endif %}",
 }


--- a/src/axolotl/utils/config/models/input/v0_4_1/init.py
+++ b/src/axolotl/utils/config/models/input/v0_4_1/init.py
@@ -666,6 +666,8 @@ class AxolotlInputConfig(
    loss_watchdog_threshold: Optional[float] = None
    loss_watchdog_patience: Optional[int] = None

+    gc_steps: Optional[int] = None
+
    bf16: Optional[Union[Literal["auto"], bool]] = "auto"
    fp16: Optional[bool] = None
    bfloat16: Optional[bool] = None  # for non-AMP cases
@@ -792,7 +794,7 @@ class AxolotlInputConfig(
    chat_template_jinja: Optional[str] = None
    default_system_message: Optional[str] = None

-    fix_untrained_tokens: Optional[bool] = None
+    fix_untrained_tokens: Optional[Union[int, List[int]]] = None

    # INTERNALS - document for now, generally not set externally
    is_preprocess: Optional[bool] = None
@@ -1627,3 +1629,19 @@ class AxolotlConfigWCapabilities(AxolotlInputConfig):
            else:
                data["torch_compile"] = False
        return data
+
+    @model_validator(mode="before")
+    @classmethod
+    def check_hymba_torch_version(cls, data):
+        if "hymba" in data.get("base_model", {}).lower():
+            env_capabilities = data.get("env_capabilities", {})
+            torch_version = env_capabilities.get("torch_version")
+
+            if torch_version is None:
+                import torch
+
+                torch_version = str(torch.__version__).split("+", maxsplit=1)[0]
+
+            if version.parse(torch_version) < version.parse("2.5.0"):
+                raise ValueError("Hymba requires torch version >= 2.5")
+        return data
--- a/src/axolotl/utils/data/pretraining.py
+++ b/src/axolotl/utils/data/pretraining.py
@@ -28,8 +28,10 @@ def encode_pretraining(
    )
    # Convert to PyTorch tensors
    input_ids = [torch.tensor(seq) for seq in res["input_ids"]]
+    targets = [torch.tensor(seq) for seq in res["input_ids"]]
    attention_mask = [torch.tensor(seq) for seq in res["attention_mask"]]
    new_input_ids = []
+    new_labels = []
    new_attention_mask = []
    # Append EOS and PAD tokens to input_ids, and correct attention_mask
    for i, _ in enumerate(input_ids):
@@ -40,22 +42,34 @@ def encode_pretraining(
            ),
            dim=0,
        )
+        targets[i] = torch.cat(
+            (
+                targets[i],
+                torch.tensor([tokenizer.eos_token_id, -100]),
+            ),
+            dim=0,
+        )
        attention_mask[i] = torch.cat((attention_mask[i], torch.tensor([1, 0])), dim=0)

    # Concatenate tokens so that their lengths are less than max_tokens
    buffer_input_ids = torch.tensor([], dtype=torch.long)
+    buffer_labels = torch.tensor([], dtype=torch.long)
    buffer_attention_mask = torch.tensor([], dtype=torch.long)

-    for ids, mask in zip(input_ids, attention_mask):
+    for ids, labels, mask in zip(input_ids, targets, attention_mask):
        if buffer_input_ids.numel() == max_tokens:
            new_input_ids.append(buffer_input_ids)
+            new_labels.append(buffer_labels)
            new_attention_mask.append(buffer_attention_mask)
            buffer_input_ids = torch.tensor([], dtype=torch.long)
+            buffer_labels = torch.tensor([], dtype=torch.long)
            buffer_attention_mask = torch.tensor([], dtype=torch.long)
            buffer_input_ids = torch.cat((buffer_input_ids, ids), dim=0)
+            buffer_labels = torch.cat((buffer_labels, labels), dim=0)
            buffer_attention_mask = torch.cat((buffer_attention_mask, mask), dim=0)
        elif buffer_input_ids.numel() + ids.numel() <= max_tokens:
            buffer_input_ids = torch.cat((buffer_input_ids, ids), dim=0)
+            buffer_labels = torch.cat((buffer_labels, labels), dim=0)
            buffer_attention_mask = torch.cat((buffer_attention_mask, mask), dim=0)
        else:
            buffer_input_ids = torch.cat(
@@ -69,6 +83,17 @@ def encode_pretraining(
                ),
                dim=0,
            )
+            buffer_labels = torch.cat(
+                (
+                    buffer_labels,
+                    torch.full(
+                        (max_tokens - buffer_labels.numel(),),
+                        -100,
+                        dtype=torch.long,
+                    ),
+                ),
+                dim=0,
+            )
            buffer_attention_mask = torch.cat(
                (
                    buffer_attention_mask,
@@ -81,11 +106,14 @@ def encode_pretraining(
                dim=0,
            )
            new_input_ids.append(buffer_input_ids)
+            new_labels.append(buffer_labels)
            new_attention_mask.append(buffer_attention_mask)
            buffer_input_ids = torch.tensor([], dtype=torch.long)
+            buffer_labels = torch.tensor([], dtype=torch.long)
            buffer_attention_mask = torch.tensor([], dtype=torch.long)

            buffer_input_ids = torch.cat((buffer_input_ids, ids), dim=0)
+            buffer_labels = torch.cat((buffer_labels, labels), dim=0)
            buffer_attention_mask = torch.cat((buffer_attention_mask, mask), dim=0)

    if buffer_input_ids.numel() > 0:  # for any leftover tokens
@@ -101,6 +129,17 @@ def encode_pretraining(
                ),
                dim=0,
            )
+            buffer_labels = torch.cat(
+                (
+                    buffer_labels,
+                    torch.full(
+                        (max_tokens - buffer_labels.numel(),),
+                        -100,
+                        dtype=torch.long,
+                    ),
+                ),
+                dim=0,
+            )
            buffer_attention_mask = torch.cat(
                (
                    buffer_attention_mask,
@@ -113,11 +152,12 @@ def encode_pretraining(
                dim=0,
            )
        new_input_ids.append(buffer_input_ids)
+        new_labels.append(buffer_labels)
        new_attention_mask.append(buffer_attention_mask)

    ret = {
        "input_ids": [seq.tolist() for seq in new_input_ids],
-        "labels": [seq.tolist() for seq in new_input_ids],
+        "labels": [seq.tolist() for seq in new_labels],
        "attention_mask": [seq.tolist() for seq in new_attention_mask],
    }

--- a/src/axolotl/utils/data/sft.py
+++ b/src/axolotl/utils/data/sft.py
@@ -3,7 +3,7 @@
 import functools
 import logging
 from pathlib import Path
-from typing import List, Optional, Tuple, Union
+from typing import List, Tuple, Union

 from datasets import (
    Dataset,
@@ -12,8 +12,6 @@ from datasets import (
    load_dataset,
    load_from_disk,
 )
-from huggingface_hub import hf_hub_download
-from huggingface_hub.utils import HFValidationError
 from transformers import PreTrainedTokenizerBase

 from axolotl.common.const import DEFAULT_DATASET_PREPARED_PATH
@@ -42,6 +40,7 @@ from axolotl.prompters import (
    UnsupportedPrompter,
 )
 from axolotl.utils.data.pretraining import wrap_pretraining_dataset
+from axolotl.utils.data.shared import load_dataset_w_config
 from axolotl.utils.data.utils import (
    deduplicate_and_log_datasets,
    md5,
@@ -85,6 +84,7 @@ def prepare_dataset(cfg, tokenizer, processor=None):
                    processor=processor,
                )
    else:
+        # Load streaming dataset if pretraining_dataset is given
        path = cfg.pretraining_dataset
        split = "train"
        name = None
@@ -116,7 +116,18 @@ def prepare_dataset(cfg, tokenizer, processor=None):
        )
        # https://discuss.huggingface.co/t/how-to-use-huggingface-trainer-streaming-datasets-without-wrapping-it-with-torchdatas-iterablewrapper/25230
        train_dataset = train_dataset.with_format("torch")
+
+        # Load eval dataset (non-streaming) if specified
        eval_dataset = None
+        if cfg.test_datasets:
+            _, eval_dataset, _ = load_prepare_datasets(
+                tokenizer,
+                cfg,
+                DEFAULT_DATASET_PREPARED_PATH,
+                split="test",
+                processor=processor,
+            )
+
        if cfg.dataset_exact_deduplication:
            LOG.info("Deduplication not available for pretrained datasets")

@@ -243,195 +254,9 @@ def load_tokenized_prepared_datasets(

        # pylint: disable=invalid-name
        for config_dataset in for_d_in_datasets(cfg_datasets):
-            ds: Optional[Union[Dataset, DatasetDict]] = None
-            ds_from_hub = False
-            ds_trust_remote_code = config_dataset.trust_remote_code
-            try:
-                # this is just a basic check to see if the path is a
-                # valid HF dataset that's loadable
-                load_dataset(
-                    config_dataset.path,
-                    name=config_dataset.name,
-                    streaming=True,
-                    token=use_auth_token,
-                    revision=config_dataset.revision,
-                    trust_remote_code=ds_trust_remote_code,
-                )
-                ds_from_hub = True
-            except (FileNotFoundError, ConnectionError, HFValidationError, ValueError):
-                pass
-
-            ds_from_cloud = False
-            storage_options = {}
-            remote_file_system = None
-            if config_dataset.path.startswith("s3://"):
-                try:
-                    import aiobotocore.session  # type: ignore
-                    import s3fs  # type: ignore
-                except ImportError as exc:
-                    raise ImportError(
-                        "s3:// paths require aiobotocore and s3fs to be installed"
-                    ) from exc
-
-                # Takes credentials from ~/.aws/credentials for default profile
-                s3_session = aiobotocore.session.AioSession(profile="default")
-                storage_options = {"session": s3_session}
-                remote_file_system = s3fs.S3FileSystem(**storage_options)
-            elif config_dataset.path.startswith(
-                "gs://"
-            ) or config_dataset.path.startswith("gcs://"):
-                try:
-                    import gcsfs  # type: ignore
-                except ImportError as exc:
-                    raise ImportError(
-                        "gs:// or gcs:// paths require gcsfs to be installed"
-                    ) from exc
-
-                # gcsfs will use default credentials from the environment else anon
-                # https://gcsfs.readthedocs.io/en/latest/#credentials
-                storage_options = {"token": None}
-                remote_file_system = gcsfs.GCSFileSystem(**storage_options)
-            # TODO: Figure out how to get auth creds passed
-            # elif config_dataset.path.startswith("adl://") or config_dataset.path.startswith("abfs://"):
-            #     try:
-            #         import adlfs
-            #     except ImportError as exc:
-            #        raise ImportError(
-            #            "adl:// or abfs:// paths require adlfs to be installed"
-            #        ) from exc
-
-            #     # Gen 1
-            #     storage_options = {
-            #         "tenant_id": TENANT_ID,
-            #         "client_id": CLIENT_ID,
-            #         "client_secret": CLIENT_SECRET,
-            #     }
-            #     # Gen 2
-            #     storage_options = {
-            #         "account_name": ACCOUNT_NAME,
-            #         "account_key": ACCOUNT_KEY,
-            #     }
-
-            #     remote_file_system = adlfs.AzureBlobFileSystem(**storage_options)
-            try:
-                if remote_file_system and remote_file_system.exists(
-                    config_dataset.path
-                ):
-                    ds_from_cloud = True
-            except (FileNotFoundError, ConnectionError):
-                pass
-
-            # prefer local dataset, even if hub exists
-            local_path = Path(config_dataset.path)
-            if local_path.exists():
-                if local_path.is_dir():
-                    if config_dataset.data_files:
-                        ds_type = get_ds_type(config_dataset)
-                        ds = load_dataset(
-                            ds_type,
-                            name=config_dataset.name,
-                            data_files=config_dataset.data_files,
-                            streaming=False,
-                            split=None,
-                        )
-                    else:
-                        try:
-                            ds = load_from_disk(config_dataset.path)
-                        except FileNotFoundError:
-                            ds = load_dataset(
-                                config_dataset.path,
-                                name=config_dataset.name,
-                                streaming=False,
-                                split=None,
-                            )
-                elif local_path.is_file():
-                    ds_type = get_ds_type(config_dataset)
-
-                    ds = load_dataset(
-                        ds_type,
-                        name=config_dataset.name,
-                        data_files=config_dataset.path,
-                        streaming=False,
-                        split=None,
-                    )
-                else:
-                    raise ValueError(
-                        "unhandled dataset load: local path exists, but is neither a directory or a file"
-                    )
-            elif ds_from_hub:
-                load_ds_kwargs = {}
-                if config_dataset.split:
-                    load_ds_kwargs["split"] = config_dataset.split
-                ds = load_dataset(
-                    config_dataset.path,
-                    name=config_dataset.name,
-                    streaming=False,
-                    data_files=config_dataset.data_files,
-                    token=use_auth_token,
-                    revision=config_dataset.revision,
-                    trust_remote_code=config_dataset.trust_remote_code,
-                    **load_ds_kwargs,
-                )
-            elif ds_from_cloud and remote_file_system:
-                if remote_file_system.isdir(config_dataset.path):
-                    ds = load_from_disk(
-                        config_dataset.path,
-                        storage_options=storage_options,
-                    )
-                elif remote_file_system.isfile(config_dataset.path):
-                    ds_type = get_ds_type(config_dataset)
-                    ds = load_dataset(
-                        ds_type,
-                        name=config_dataset.name,
-                        data_files=config_dataset.path,
-                        streaming=False,
-                        split=None,
-                        storage_options=storage_options,
-                        trust_remote_code=config_dataset.trust_remote_code,
-                    )
-            elif config_dataset.path.startswith("https://"):
-                ds_type = get_ds_type(config_dataset)
-                ds = load_dataset(
-                    ds_type,
-                    name=config_dataset.name,
-                    data_files=config_dataset.path,
-                    streaming=False,
-                    split=None,
-                    storage_options=storage_options,
-                    trust_remote_code=config_dataset.trust_remote_code,
-                )
-            else:
-                if isinstance(config_dataset.data_files, str):
-                    fp = hf_hub_download(
-                        repo_id=config_dataset.path,
-                        repo_type="dataset",
-                        filename=config_dataset.data_files,
-                        revision=config_dataset.revision,
-                    )
-                elif isinstance(config_dataset.data_files, list):
-                    fp = []
-                    for file in config_dataset.data_files:
-                        fp.append(
-                            hf_hub_download(
-                                repo_id=config_dataset.path,
-                                repo_type="dataset",
-                                filename=file,
-                                revision=config_dataset.revision,
-                            )
-                        )
-                else:
-                    raise ValueError(
-                        "data_files must be either a string or list of strings"
-                    )
-                ds = load_dataset(
-                    "json",
-                    name=config_dataset.name,
-                    data_files=fp,
-                    streaming=False,
-                    split=None,
-                )
-            if not ds:
-                raise ValueError("unhandled dataset load")
+            ds: Union[Dataset, DatasetDict] = load_dataset_w_config(
+                config_dataset, use_auth_token
+            )

            d_base_type = d_prompt_style = None
            d_type = config_dataset.type
@@ -501,24 +326,6 @@ def load_tokenized_prepared_datasets(
    return dataset, prompters


-def get_ds_type(config_dataset: DictDefault):
-    """
-    Get the dataset type from the path if it's not specified
-    """
-    ds_type = "json"
-    if config_dataset.ds_type:
-        ds_type = config_dataset.ds_type
-    elif ".parquet" in config_dataset.path:
-        ds_type = "parquet"
-    elif ".arrow" in config_dataset.path:
-        ds_type = "arrow"
-    elif ".csv" in config_dataset.path:
-        ds_type = "csv"
-    elif ".txt" in config_dataset.path:
-        ds_type = "text"
-    return ds_type
-
-
 def load_prepare_datasets(
    tokenizer: PreTrainedTokenizerBase,
    cfg,
--- a/src/axolotl/utils/data/shared.py
+++ b/src/axolotl/utils/data/shared.py
@@ -0,0 +1,222 @@
+"""
+dataset loading shared utils
+"""
+from pathlib import Path
+from typing import Optional, Union
+
+from datasets import Dataset, DatasetDict, load_dataset, load_from_disk
+from huggingface_hub import hf_hub_download
+from huggingface_hub.errors import HFValidationError
+
+from axolotl.utils.dict import DictDefault
+
+
+def get_ds_type(config_dataset: DictDefault):
+    """
+    Get the dataset type from the path if it's not specified
+    """
+    ds_type = "json"
+    if config_dataset.ds_type:
+        ds_type = config_dataset.ds_type
+    elif ".parquet" in config_dataset.path:
+        ds_type = "parquet"
+    elif ".arrow" in config_dataset.path:
+        ds_type = "arrow"
+    elif ".csv" in config_dataset.path:
+        ds_type = "csv"
+    elif ".txt" in config_dataset.path:
+        ds_type = "text"
+    return ds_type
+
+
+def load_dataset_w_config(config_dataset, auth_token):
+    # pylint: disable=invalid-name
+    ds: Optional[Union[Dataset, DatasetDict]] = None  # pylint: disable=invalid-name
+    ds_from_hub = False
+    ds_trust_remote_code = config_dataset.trust_remote_code
+    try:
+        # this is just a basic check to see if the path is a
+        # valid HF dataset that's loadable
+        load_dataset(
+            config_dataset.path,
+            name=config_dataset.name,
+            streaming=True,
+            token=auth_token,
+            revision=config_dataset.revision,
+            trust_remote_code=ds_trust_remote_code,
+        )
+        ds_from_hub = True
+    except (FileNotFoundError, ConnectionError, HFValidationError, ValueError):
+        pass
+
+    ds_from_cloud = False
+    storage_options = {}
+    remote_file_system = None
+    if config_dataset.path.startswith("s3://"):
+        try:
+            import aiobotocore.session  # type: ignore
+            import s3fs  # type: ignore
+        except ImportError as exc:
+            raise ImportError(
+                "s3:// paths require aiobotocore and s3fs to be installed"
+            ) from exc
+
+        # Takes credentials from ~/.aws/credentials for default profile
+        s3_session = aiobotocore.session.AioSession(profile="default")
+        storage_options = {"session": s3_session}
+        remote_file_system = s3fs.S3FileSystem(**storage_options)
+    elif config_dataset.path.startswith("gs://") or config_dataset.path.startswith(
+        "gcs://"
+    ):
+        try:
+            import gcsfs  # type: ignore
+        except ImportError as exc:
+            raise ImportError(
+                "gs:// or gcs:// paths require gcsfs to be installed"
+            ) from exc
+
+        # gcsfs will use default credentials from the environment else anon
+        # https://gcsfs.readthedocs.io/en/latest/#credentials
+        storage_options = {"token": None}
+        remote_file_system = gcsfs.GCSFileSystem(**storage_options)
+    # TODO: Figure out how to get auth creds passed
+    # elif config_dataset.path.startswith("adl://") or config_dataset.path.startswith("abfs://"):
+    #     try:
+    #         import adlfs
+    #     except ImportError as exc:
+    #        raise ImportError(
+    #            "adl:// or abfs:// paths require adlfs to be installed"
+    #        ) from exc
+
+    #     # Gen 1
+    #     storage_options = {
+    #         "tenant_id": TENANT_ID,
+    #         "client_id": CLIENT_ID,
+    #         "client_secret": CLIENT_SECRET,
+    #     }
+    #     # Gen 2
+    #     storage_options = {
+    #         "account_name": ACCOUNT_NAME,
+    #         "account_key": ACCOUNT_KEY,
+    #     }
+
+    #     remote_file_system = adlfs.AzureBlobFileSystem(**storage_options)
+    try:
+        if remote_file_system and remote_file_system.exists(config_dataset.path):
+            ds_from_cloud = True
+    except (FileNotFoundError, ConnectionError):
+        pass
+
+    # prefer local dataset, even if hub exists
+    local_path = Path(config_dataset.path)
+    if local_path.exists():
+        if local_path.is_dir():
+            if config_dataset.data_files:
+                ds_type = get_ds_type(config_dataset)
+                ds = load_dataset(  # pylint: disable=invalid-name
+                    ds_type,
+                    name=config_dataset.name,
+                    data_files=config_dataset.data_files,
+                    streaming=False,
+                    split=None,
+                )
+            else:
+                try:
+                    ds = load_from_disk(
+                        config_dataset.path
+                    )  # pylint: disable=invalid-name
+                except FileNotFoundError:
+                    ds = load_dataset(
+                        config_dataset.path,
+                        name=config_dataset.name,
+                        streaming=False,
+                        split=None,
+                    )
+        elif local_path.is_file():
+            ds_type = get_ds_type(config_dataset)
+
+            ds = load_dataset(  # pylint: disable=invalid-name
+                ds_type,
+                name=config_dataset.name,
+                data_files=config_dataset.path,
+                streaming=False,
+                split=None,
+            )
+        else:
+            raise ValueError(
+                "unhandled dataset load: local path exists, but is neither a directory or a file"
+            )
+    elif ds_from_hub:
+        load_ds_kwargs = {}
+        if config_dataset.split:
+            load_ds_kwargs["split"] = config_dataset.split
+        ds = load_dataset(
+            config_dataset.path,
+            name=config_dataset.name,
+            streaming=False,
+            data_files=config_dataset.data_files,
+            token=auth_token,
+            revision=config_dataset.revision,
+            trust_remote_code=config_dataset.trust_remote_code,
+            **load_ds_kwargs,
+        )
+    elif ds_from_cloud and remote_file_system:
+        if remote_file_system.isdir(config_dataset.path):
+            ds = load_from_disk(
+                config_dataset.path,
+                storage_options=storage_options,
+            )
+        elif remote_file_system.isfile(config_dataset.path):
+            ds_type = get_ds_type(config_dataset)
+            ds = load_dataset(
+                ds_type,
+                name=config_dataset.name,
+                data_files=config_dataset.path,
+                streaming=False,
+                split=None,
+                storage_options=storage_options,
+                trust_remote_code=config_dataset.trust_remote_code,
+            )
+    elif config_dataset.path.startswith("https://"):
+        ds_type = get_ds_type(config_dataset)
+        ds = load_dataset(
+            ds_type,
+            name=config_dataset.name,
+            data_files=config_dataset.path,
+            streaming=False,
+            split=None,
+            storage_options=storage_options,
+            trust_remote_code=config_dataset.trust_remote_code,
+        )
+    else:
+        if isinstance(config_dataset.data_files, str):
+            fp = hf_hub_download(
+                repo_id=config_dataset.path,
+                repo_type="dataset",
+                filename=config_dataset.data_files,
+                revision=config_dataset.revision,
+            )
+        elif isinstance(config_dataset.data_files, list):
+            fp = []
+            for file in config_dataset.data_files:
+                fp.append(
+                    hf_hub_download(
+                        repo_id=config_dataset.path,
+                        repo_type="dataset",
+                        filename=file,
+                        revision=config_dataset.revision,
+                    )
+                )
+        else:
+            raise ValueError("data_files must be either a string or list of strings")
+        ds = load_dataset(
+            "json",
+            name=config_dataset.name,
+            data_files=fp,
+            streaming=False,
+            split=None,
+        )
+    if not ds:
+        raise ValueError("unhandled dataset load")
+
+    return ds
--- a/src/axolotl/utils/models.py
+++ b/src/axolotl/utils/models.py
@@ -409,6 +409,7 @@ class ModelLoader:
            and self.cfg.sample_packing
        ):
            if "auto_map" in self.model_config:
+                # some model config objects are not subscriptable
                try:
                    auto_map_config = self.model_config["auto_map"]
                except TypeError:
--- a/tests/e2e/test_optimizers.py
+++ b/tests/e2e/test_optimizers.py
@@ -67,8 +67,8 @@ class TestCustomOptimizers(unittest.TestCase):
        train(cfg=cfg, cli_args=cli_args, dataset_meta=dataset_meta)
        assert (Path(temp_dir) / "adapter_model.bin").exists()

-    @with_temp_dir
    @require_torch_2_5_1
+    @with_temp_dir
    def test_adopt_adamw(self, temp_dir):
        # pylint: disable=duplicate-code
        cfg = DictDefault(
--- a/tests/e2e/test_packing_loss.py
+++ b/tests/e2e/test_packing_loss.py
@@ -14,7 +14,7 @@ from axolotl.train import train
 from axolotl.utils.config import normalize_config
 from axolotl.utils.dict import DictDefault

-from .utils import check_tensorboard, with_temp_dir
+from .utils import check_tensorboard, require_torch_2_5_1, with_temp_dir

 LOG = logging.getLogger("axolotl.tests.e2e")
 os.environ["WANDB_DISABLED"] = "true"
@@ -68,3 +68,129 @@ class TestPackedLlama(unittest.TestCase):
        check_tensorboard(
            temp_dir + "/runs", "train/train_loss", 2.0, "Train Loss is too high"
        )
+
+
+class TestUnpackedHymba(unittest.TestCase):
+    """
+    Test case for Unpacked training of hymba models
+    """
+
+    @require_torch_2_5_1
+    @with_temp_dir
+    def test_loss_unpacked(self, temp_dir):
+        # pylint: disable=duplicate-code
+        cfg = DictDefault(
+            {
+                "base_model": "nvidia/Hymba-1.5B-Base",
+                "trust_remote_code": True,
+                "load_in_4bit": True,
+                "adapter": "qlora",
+                "lora_r": 32,
+                "lora_alpha": 16,
+                "lora_dropout": 0.05,
+                "lora_target_modules": [
+                    "gate_proj",
+                    "down_proj",
+                    "up_proj",
+                    "q_proj",
+                    "v_proj",
+                    "k_proj",
+                    "o_proj",
+                ],
+                "sequence_len": 1024,
+                "sample_packing": False,
+                "flash_attention": True,
+                "val_set_size": 0.0,
+                "datasets": [
+                    {
+                        "path": "vicgalle/alpaca-gpt4",
+                        "type": "alpaca",
+                    },
+                ],
+                "num_epochs": 1,
+                "micro_batch_size": 2,
+                "gradient_accumulation_steps": 4,
+                "output_dir": temp_dir,
+                "learning_rate": 0.00001,
+                "optimizer": "adamw_torch",
+                "lr_scheduler": "cosine",
+                "max_steps": 5,
+                "use_tensorboard": True,
+            }
+        )
+        if is_torch_bf16_gpu_available():
+            cfg.bf16 = True
+        else:
+            cfg.fp16 = True
+        normalize_config(cfg)
+        cli_args = TrainerCliArgs()
+        dataset_meta = load_datasets(cfg=cfg, cli_args=cli_args)
+
+        train(cfg=cfg, cli_args=cli_args, dataset_meta=dataset_meta)
+
+        check_tensorboard(
+            temp_dir + "/runs", "train/train_loss", 2.0, "Train Loss is too high"
+        )
+
+
+class TestPackedHymba(unittest.TestCase):
+    """
+    Test case for Packed training of hymba models
+    """
+
+    @require_torch_2_5_1
+    @with_temp_dir
+    def test_loss_packed(self, temp_dir):
+        # pylint: disable=duplicate-code
+        cfg = DictDefault(
+            {
+                "base_model": "nvidia/Hymba-1.5B-Base",
+                "trust_remote_code": True,
+                "load_in_4bit": True,
+                "adapter": "qlora",
+                "lora_r": 32,
+                "lora_alpha": 16,
+                "lora_dropout": 0.05,
+                "lora_target_modules": [
+                    "gate_proj",
+                    "down_proj",
+                    "up_proj",
+                    "q_proj",
+                    "v_proj",
+                    "k_proj",
+                    "o_proj",
+                ],
+                "sequence_len": 1024,
+                "sample_packing": True,
+                "flash_attention": True,
+                "val_set_size": 0.0,
+                "datasets": [
+                    {
+                        "path": "vicgalle/alpaca-gpt4",
+                        "type": "alpaca",
+                    },
+                ],
+                "num_epochs": 1,
+                "micro_batch_size": 2,
+                "gradient_accumulation_steps": 4,
+                "output_dir": temp_dir,
+                "learning_rate": 0.00001,
+                "optimizer": "adamw_torch",
+                "lr_scheduler": "cosine",
+                "max_steps": 5,
+                "use_tensorboard": True,
+            }
+        )
+        if is_torch_bf16_gpu_available():
+            cfg.bf16 = True
+        else:
+            cfg.fp16 = True
+        normalize_config(cfg)
+        cli_args = TrainerCliArgs()
+        dataset_meta = load_datasets(cfg=cfg, cli_args=cli_args)
+
+        train(cfg=cfg, cli_args=cli_args, dataset_meta=dataset_meta)
+
+        check_tensorboard(
+            temp_dir + "/runs", "train/train_loss", 2.0, "Train Loss is too high"
+        )
Author	SHA1	Message	Date
Sunny Liu	e7912a4a66	Merge branch 'main' into hymba_multipack2	2025-01-05 23:15:57 -05:00
Wing Lian	3915abee4c	make sure padding is labeled as -100 for pretraining (#2227 )	2024-12-31 15:22:18 -05:00
NJordan72	7a38dbe674	fix: allow trainer builder to use custom jinja chat template (#2219 ) * fix: allow trainer builder to use custom jinja chat template * chore: use get_chat_template_from_config Co-authored-by: Chirag Jain <jain.chirag925@gmail.com> * fix: swap imports --------- Co-authored-by: Chirag Jain <jain.chirag925@gmail.com>	2024-12-24 16:18:50 -05:00
Wing Lian	e0a2eb2ebd	fix untrained tokens if specified explicitly from a list (#2210 )	2024-12-23 09:08:28 -05:00
Wing Lian	d852d7af7a	inference - don't default w accelerate, fix base model (#2216 ) [skip ci]	2024-12-23 07:48:41 -05:00
Wing Lian	3742deb1de	add deepspeed example with torch compile enabled (#2212 ) [skip ci]	2024-12-22 12:11:39 -05:00
Wing Lian	2312caaa98	GC every n steps (#2209 )	2024-12-21 17:38:33 -05:00
Wing Lian	307cf7c685	move the dataset loading from remote/disk to a shared function so we can re-use for RL (#2204 )	2024-12-20 21:43:52 -05:00
Dan Saunders	70541145f1	adding test_datasets compat with pretraining_dataset (streaming) (#2206 ) [skip ci]	2024-12-20 21:43:33 -05:00
bursteratom	26cd287cab	switching test hymba order	2024-12-19 20:42:52 -05:00
bursteratom	cce7007bf8	rebased hymba multipack	2024-12-19 20:42:52 -05:00
Wing Lian	42bd32a233	add outputs (symlink) to gitignore [skip ci] (#2205 )	2024-12-19 20:14:43 -05:00
Dan Saunders	5b8fb5e939	remove cicd pytest xdist args (#2201 ) * remove cicd pytest xdist args * Delete outputs	2024-12-19 11:44:53 -05:00