From 50682a3c068f723de154950b03c3f86bf673e688 Mon Sep 17 00:00:00 2001 From: Wing Lian Date: Mon, 21 Aug 2023 16:43:33 -0400 Subject: [PATCH] always drop samples that are too long (#452) --- src/axolotl/utils/trainer.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/src/axolotl/utils/trainer.py b/src/axolotl/utils/trainer.py index 4cf092ecd..c9c17fe33 100644 --- a/src/axolotl/utils/trainer.py +++ b/src/axolotl/utils/trainer.py @@ -284,15 +284,15 @@ def disable_datasets_caching(): def process_datasets_for_packing(cfg, train_dataset, eval_dataset): + drop_long = partial(drop_long_seq, sequence_len=cfg.sequence_len) + train_dataset = train_dataset.filter(drop_long, num_proc=os.cpu_count()) + if eval_dataset: + eval_dataset = eval_dataset.filter(drop_long, num_proc=os.cpu_count()) + if cfg.sample_packing: - drop_long = partial(drop_long_seq, sequence_len=cfg.sequence_len) - train_dataset = train_dataset.filter(drop_long, num_proc=os.cpu_count()).map( - add_position_ids, num_proc=os.cpu_count() - ) + train_dataset = train_dataset.map(add_position_ids, num_proc=os.cpu_count()) if eval_dataset: - eval_dataset = eval_dataset.filter(drop_long, num_proc=os.cpu_count()).map( - add_position_ids, num_proc=os.cpu_count() - ) + eval_dataset = eval_dataset.map(add_position_ids, num_proc=os.cpu_count()) return train_dataset, eval_dataset