various tests fixes for flakey tests (#2110)

* add mhenrichsen/alpaca_2k_test with revision dataset download fixture for flaky tests * log slowest tests * pin pynvml==11.5.3 * fix load local hub path * optimize for speed w smaller models and val_set_size * replace pynvml * make the resume from checkpoint e2e faster * make tests smaller
2024-12-02 17:28:58 -05:00
parent b620ed94d0
commit ce5bcff750
13 changed files with 78 additions and 44 deletions
--- a/src/axolotl/utils/bench.py
+++ b/src/axolotl/utils/bench.py
@@ -1,13 +1,24 @@
 """Benchmarking and measurement utilities"""
 import functools

-import pynvml
 import torch
-from pynvml.nvml import NVMLError
 from transformers.utils.import_utils import is_torch_npu_available

 from axolotl.utils.distributed import get_device_type

+try:
+    from pynvml import (
+        NVMLError,
+        nvmlDeviceGetHandleByIndex,
+        nvmlDeviceGetMemoryInfo,
+        nvmlInit,
+    )
+except ImportError:
+    NVMLError = None
+    nvmlDeviceGetHandleByIndex = None
+    nvmlDeviceGetMemoryInfo = None
+    nvmlInit = None
+

 def check_cuda_device(default_value):
    """
@@ -68,10 +79,12 @@ def gpu_memory_usage_smi(device=0):
        device = device.index
    if isinstance(device, str) and device.startswith("cuda:"):
        device = int(device[5:])
+    if not nvmlInit:
+        return 0.0
    try:
-        pynvml.nvmlInit()
-        handle = pynvml.nvmlDeviceGetHandleByIndex(device)
-        info = pynvml.nvmlDeviceGetMemoryInfo(handle)
+        nvmlInit()
+        handle = nvmlDeviceGetHandleByIndex(device)
+        info = nvmlDeviceGetMemoryInfo(handle)
        return info.used / 1024.0**3
    except NVMLError:
        return 0.0
--- a/src/axolotl/utils/data/sft.py
+++ b/src/axolotl/utils/data/sft.py
@@ -179,7 +179,7 @@ def load_tokenized_prepared_datasets(
                + "|".join(
                    sorted(
                        [
-                            f"{d.path}: {d.type}: {d.shards}: {d.conversation}{d.split}"
+                            f"{d.path}:{d.type}:{d.shards}:{d.conversation}{d.split}"
                            for d in cfg_datasets
                        ]
                    )