From de407b59ea329923b9b33177bc3d11186a67f38f Mon Sep 17 00:00:00 2001 From: hiyouga Date: Wed, 2 Aug 2023 01:10:28 +0800 Subject: [PATCH] fix bug in preprocessing Former-commit-id: 968ce0dcce6bfef582ce37aea6566a65f5aac811 --- src/llmtuner/dsets/preprocess.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/llmtuner/dsets/preprocess.py b/src/llmtuner/dsets/preprocess.py index 10c76f2b..0257b244 100644 --- a/src/llmtuner/dsets/preprocess.py +++ b/src/llmtuner/dsets/preprocess.py @@ -25,8 +25,8 @@ def preprocess_dataset( for i in range(len(examples["prompt"])): query, response = examples["prompt"][i], examples["response"][i] query = query + "\n" + examples["query"][i] if "query" in examples and examples["query"][i] else query - history = history if "history" in examples and examples["history"][i] else [] - prefix = prefix if "prefix" in examples and examples["prefix"][i] else "" + history = examples["history"][i] if "history" in examples else None + prefix = examples["prefix"][i] if "prefix" in examples else None yield query, response, history, prefix def preprocess_pretrain_dataset(examples: Dict[str, List[Any]]) -> Dict[str, Any]: