From 1162b93b6bbaf5d52f83d0a31ba0918f74eb390a Mon Sep 17 00:00:00 2001 From: Wing Lian Date: Tue, 8 Aug 2023 00:50:56 -0400 Subject: [PATCH] filter w multiple cpus --- src/axolotl/utils/trainer.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/axolotl/utils/trainer.py b/src/axolotl/utils/trainer.py index fd14d5cbf..51f41168c 100644 --- a/src/axolotl/utils/trainer.py +++ b/src/axolotl/utils/trainer.py @@ -273,11 +273,11 @@ def disable_datasets_caching(): def process_datasets_for_packing(cfg, train_dataset, eval_dataset): if cfg.sample_packing: drop_long = partial(drop_long_seq, sequence_len=cfg.sequence_len) - train_dataset = train_dataset.filter(drop_long).map( + train_dataset = train_dataset.filter(drop_long, num_proc=os.cpu_count()).map( add_position_ids, num_proc=os.cpu_count() ) if eval_dataset: - eval_dataset = eval_dataset.filter(drop_long).map( + eval_dataset = eval_dataset.filter(drop_long, num_proc=os.cpu_count()).map( add_position_ids, num_proc=os.cpu_count() ) return train_dataset, eval_dataset