refactor model_dtype, fix PPO trainer

2025-12-15 19:30:36 +08:00 · 2023-10-11 23:16:01 +08:00
parent 5310e4d182
commit 2818af0b09
10 changed files with 104 additions and 119 deletions
--- a/src/llmtuner/webui/runner.py
+++ b/src/llmtuner/webui/runner.py
@@ -145,6 +145,9 @@ class Runner:
        )
        args[compute_type] = True

+        if args["quantization_bit"] is not None:
+            args["upcast_layernorm"] = True
+
        if args["stage"] == "ppo":
            args["reward_model"] = reward_model
            val_size = 0