[test] add npu test yaml and add ascend a3 docker file (#9547)

Co-authored-by: jiaqiw09 <jiaqiw960714@gmail.com>
2026-07-31 05:06:10 +08:00 · 2025-11-30 09:37:08 +08:00
parent 22be45c78c
commit e43a972b25
33 changed files with 322 additions and 21 deletions
--- a/tests/model/test_lora.py
+++ b/tests/model/test_lora.py
@@ -55,30 +55,35 @@ INFER_ARGS = {
 }


+@pytest.mark.runs_on(["cpu","npu"])
 def test_lora_train_qv_modules():
    model = load_train_model(lora_target="q_proj,v_proj", **TRAIN_ARGS)
    linear_modules, _ = check_lora_model(model)
    assert linear_modules == {"q_proj", "v_proj"}


+@pytest.mark.runs_on(["cpu","npu"])
 def test_lora_train_all_modules():
    model = load_train_model(lora_target="all", **TRAIN_ARGS)
    linear_modules, _ = check_lora_model(model)
    assert linear_modules == {"q_proj", "k_proj", "v_proj", "o_proj", "up_proj", "gate_proj", "down_proj"}


+@pytest.mark.runs_on(["cpu","npu"])
 def test_lora_train_extra_modules():
    model = load_train_model(additional_target="embed_tokens,lm_head", **TRAIN_ARGS)
    _, extra_modules = check_lora_model(model)
    assert extra_modules == {"embed_tokens", "lm_head"}


+@pytest.mark.runs_on(["cpu","npu"])
 def test_lora_train_old_adapters():
    model = load_train_model(adapter_name_or_path=TINY_LLAMA_ADAPTER, create_new_adapter=False, **TRAIN_ARGS)
    ref_model = load_reference_model(TINY_LLAMA3, TINY_LLAMA_ADAPTER, use_lora=True, is_trainable=True)
    compare_model(model, ref_model)


+@pytest.mark.runs_on(["cpu","npu"])
 def test_lora_train_new_adapters():
    model = load_train_model(adapter_name_or_path=TINY_LLAMA_ADAPTER, create_new_adapter=True, **TRAIN_ARGS)
    ref_model = load_reference_model(TINY_LLAMA3, TINY_LLAMA_ADAPTER, use_lora=True, is_trainable=True)
@@ -87,6 +92,7 @@ def test_lora_train_new_adapters():
    )


+@pytest.mark.runs_on(["cpu","npu"])
@pytest.mark.usefixtures("fix_valuehead_cpu_loading")
 def test_lora_train_valuehead():
    model = load_train_model(add_valuehead=True, **TRAIN_ARGS)
@@ -96,7 +102,8 @@ def test_lora_train_valuehead():
    assert torch.allclose(state_dict["v_head.summary.weight"], ref_state_dict["v_head.summary.weight"])
    assert torch.allclose(state_dict["v_head.summary.bias"], ref_state_dict["v_head.summary.bias"])

-
+@pytest.mark.runs_on(["cpu","npu"])
+@pytest.mark.skip_on_devices("npu")
 def test_lora_inference():
    model = load_infer_model(**INFER_ARGS)
    ref_model = load_reference_model(TINY_LLAMA3, TINY_LLAMA_ADAPTER, use_lora=True).merge_and_unload()