[v1] add init plugin (#9716)

2026-04-24 06:39:08 +08:00 · 2026-01-04 20:51:46 +08:00
parent 81b8a50aa5
commit f60a6e3d01
14 changed files with 307 additions and 74 deletions
--- a/src/llamafactory/v1/core/base_sampler.py
+++ b/src/llamafactory/v1/core/base_sampler.py
@@ -0,0 +1,77 @@
+# Copyright 2025 the LlamaFactory team.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from abc import ABC, abstractmethod
+
+from ..config import ModelArguments, SampleArguments, SampleBackend
+from ..utils.types import HFModel, Processor, TorchDataset
+
+
+class BaseEngine(ABC):
+    @abstractmethod
+    def __init__(
+        self,
+        args: SampleArguments,
+        model_args: ModelArguments,
+        model: HFModel = None,
+        processor: Processor = None,
+    ) -> None:
+        """Initialize the engine.
+
+        Args:
+            args: Sample arguments.
+            model_args: Model arguments.
+            model: Model.
+            processor: Processor.
+        """
+        ...
+
+    @abstractmethod
+    async def generate(self, messages):
+        pass
+
+    @abstractmethod
+    async def batch_infer(self, data: TorchDataset) -> None:
+        pass
+
+
+class HuggingFaceEngine(BaseEngine):
+    def __init__(
+        self,
+        args: SampleArguments,
+        model_args: ModelArguments,
+        model: HFModel,
+        processor: Processor,
+    ) -> None:
+        self.args = args
+
+
+class BaseSampler:
+    def __init__(
+        self,
+        args: SampleArguments,
+        model_args: ModelArguments,
+        model: HFModel,
+        processor: Processor,
+    ) -> None:
+        if args.sample_backend == SampleBackend.HF:
+            self.engine = HuggingFaceEngine(args, model_args, model, processor)
+        else:
+            raise ValueError(f"Unknown sample backend: {args.sample_backend}")
+
+    async def generate(self, messages):
+        return await self.engine.generate(messages)
+
+    async def batch_infer(self, data: TorchDataset) -> None:
+        return await self.engine.batch_infer(data)
--- a/src/llamafactory/v1/core/chat_sampler.py
+++ b/src/llamafactory/v1/core/chat_sampler.py
@@ -1,44 +0,0 @@
-# Copyright 2025 the LlamaFactory team.
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-
-from abc import ABC, abstractmethod
-
-from ..config.sample_args import SampleArguments, SampleBackend
-from .model_loader import ModelLoader
-
-
-class BaseEngine(ABC):
-    @abstractmethod
-    def __init__(self, sample_args: SampleArguments, model_loader: ModelLoader) -> None: ...
-
-    @abstractmethod
-    async def generate(self):
-        pass
-
-    @abstractmethod
-    async def batch_infer(self):
-        pass
-
-
-class HuggingFaceEngine(BaseEngine):
-    def __init__(self, model_loader: ModelLoader, sample_args: SampleArguments) -> None:
-        self.args = sample_args
-
-
-class ChatSampler:
-    def __init__(self, model_loader: ModelLoader, sample_args: SampleArguments) -> None:
-        if sample_args.sample_backend == SampleBackend.HF:
-            self.engine = HuggingFaceEngine(model_loader, sample_args)
-        else:
-            raise ValueError(f"Unknown sample backend: {sample_args.sample_backend}")
--- a/src/llamafactory/v1/core/model_loader.py
+++ b/src/llamafactory/v1/core/model_loader.py
@@ -14,17 +14,24 @@

 """The definition of model loader.

-Init Phase:
+How to use:
+model_loader = ModelLoader(model_args, is_trainable=True)
+model_loader.processor: Get the tokenizer or multi-modal processor.
+model_loader.model_config: Get the model configuration.
+model_loader.model: Get the HF model.
+
+Init Workflow:
 1. Init processor.
 2. Init model config.
 3. Init model.
 4. Init adapter.
-
 """

 import torch
+from accelerate import init_empty_weights
 from transformers import AutoConfig, AutoProcessor

+from ..accelerator.helper import DeviceType
 from ..accelerator.interface import DistributedInterface
 from ..config.model_args import ModelArguments, ModelClass
 from ..utils import logging
@@ -55,11 +62,14 @@ class ModelLoader:
        """HF model."""

    def _init_processor(self) -> Processor:
-        """Init processor."""
+        """Init processor.
+
+        NOTE: Transformers v5 always use fast tokenizer.
+        https://github.com/huggingface/transformers/blob/v5.0.0rc1/src/transformers/models/auto/tokenization_auto.py#L642
+        """
        return AutoProcessor.from_pretrained(
            self.args.model,
            trust_remote_code=self.args.trust_remote_code,
-            use_fast=self.args.use_fast_processor,
        )

    def _init_model_config(self) -> HFConfig:
@@ -92,14 +102,24 @@ class ModelLoader:

            AutoClass = AutoModel

-        # map the entire model to the current accelerator
-        model = AutoClass.from_pretrained(
-            self.args.model,
-            config=self.model_config,
-            dtype="auto",
-            device_map=DistributedInterface().current_accelerator,
-            trust_remote_code=self.args.trust_remote_code,
-        )
+        if self.args.init_config is not None:
+            from ..plugins.model_plugins.initialization import InitPlugin
+
+            init_device = InitPlugin(self.args.init_config.name)()
+        else:
+            init_device = DistributedInterface().current_accelerator
+
+        if init_device.type == DeviceType.META:
+            with init_empty_weights():
+                model = AutoClass.from_config(self.model_config)
+        else:
+            model = AutoClass.from_pretrained(
+                self.args.model,
+                config=self.model_config,
+                dtype="auto",
+                device_map=init_device,
+                trust_remote_code=self.args.trust_remote_code,
+            )

        if self.args.peft_config is None:
            if self.is_train: