fix flaky tests; should be using train loss from final step rather than final avg train loss

2026-03-22 10:38:46 -04:00
parent 5b2e3f00ce
commit 6130e40c37
21 changed files with 37 additions and 41 deletions
--- a/tests/e2e/patched/test_fa_xentropy.py
+++ b/tests/e2e/patched/test_fa_xentropy.py
@@ -78,5 +78,5 @@ class TestFAXentropyLlama:
        check_model_output_exists(temp_dir, cfg)

        check_tensorboard(
-            temp_dir + "/runs", "train/train_loss", 1.5, "Train Loss (%s) is too high"
+            temp_dir + "/runs", "train/loss", 1.5, "Train Loss (%s) is too high"
        )
--- a/tests/e2e/patched/test_flattening.py
+++ b/tests/e2e/patched/test_flattening.py
@@ -77,5 +77,5 @@ class TestFAFlattening:
        check_model_output_exists(temp_dir, cfg)

        check_tensorboard(
-            temp_dir + "/runs", "train/train_loss", 1.5, "Train Loss (%s) is too high"
+            temp_dir + "/runs", "train/loss", 1.5, "Train Loss (%s) is too high"
        )
--- a/tests/e2e/patched/test_unsloth_qlora.py
+++ b/tests/e2e/patched/test_unsloth_qlora.py
@@ -73,7 +73,7 @@ class TestUnslothQLoRA:
        check_model_output_exists(temp_dir, cfg)

        check_tensorboard(
-            temp_dir + "/runs", "train/train_loss", 2.0, "Train Loss (%s) is too high"
+            temp_dir + "/runs", "train/loss", 2.0, "Train Loss (%s) is too high"
        )

    def test_unsloth_llama_qlora_unpacked(self, temp_dir):
@@ -124,7 +124,7 @@ class TestUnslothQLoRA:
        check_model_output_exists(temp_dir, cfg)

        check_tensorboard(
-            temp_dir + "/runs", "train/train_loss", 2.0, "Train Loss (%s) is too high"
+            temp_dir + "/runs", "train/loss", 2.0, "Train Loss (%s) is too high"
        )

    @pytest.mark.parametrize(
@@ -180,5 +180,5 @@ class TestUnslothQLoRA:
        check_model_output_exists(temp_dir, cfg)

        check_tensorboard(
-            temp_dir + "/runs", "train/train_loss", 2.0, "Train Loss (%s) is too high"
+            temp_dir + "/runs", "train/loss", 2.0, "Train Loss (%s) is too high"
        )