[Device]Support npu (#6159)

* support npu * support pretrain support pretrain fix * support lora fix fix * support chatglm fix fxi fix [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci fix fix [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci fix [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci fix fix fix * Update train.py * Update train.py * fix * fix * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * fix * fix * fix * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2025-09-02 01:28:31 +00:00 · 2024-12-17 15:42:39 +08:00
parent e994c64568
commit aaafb38851
18 changed files with 295 additions and 152 deletions
--- a/applications/Colossal-LLaMA/colossal_llama/dataset/conversation.py
+++ b/applications/Colossal-LLaMA/colossal_llama/dataset/conversation.py
@@ -100,7 +100,7 @@ LLaMA3_Conv = Conversation(
    messages=[],
    offset=0,
    sep_style=SeparatorStyle.ADD_BOS_EOS_TOKEN,
-    seps=["<|begin_of_text|>", "<|end_of_text|>"],
+    seps=["<|begin_of_text|>", "<|eot_id|>"],
 )

 default_conversation = LLaMA3_Conv
--- a/applications/Colossal-LLaMA/colossal_llama/dataset/spliced_and_tokenized_dataset.py
+++ b/applications/Colossal-LLaMA/colossal_llama/dataset/spliced_and_tokenized_dataset.py
@@ -88,7 +88,7 @@ def supervised_tokenize_sft(

    assert (
        tokenizer.bos_token == conversation_template.seps[0] and tokenizer.eos_token == conversation_template.seps[1]
-    ), "`bos_token` and `eos_token` should be the same with `conversation_template.seps`."
+    ), f"`bos_token`{tokenizer.bos_token} and `eos_token`{tokenizer.eos_token} should be the same with `conversation_template.seps`{conversation_template.seps}."

    if ignore_index is None:
        ignore_index = IGNORE_INDEX
--- a/applications/Colossal-LLaMA/colossal_llama/utils/ckpt_io.py
+++ b/applications/Colossal-LLaMA/colossal_llama/utils/ckpt_io.py
@@ -43,6 +43,7 @@ def save_checkpoint(
    step: int,
    batch_size: int,
    coordinator: DistCoordinator,
+    use_lora: bool = False,
 ) -> None:
    """
    Save model checkpoint, optimizer, LR scheduler and intermedidate running states.
@@ -51,7 +52,10 @@ def save_checkpoint(
    save_dir = os.path.join(save_dir, f"epoch-{epoch}_step-{step}")
    os.makedirs(os.path.join(save_dir, "modeling"), exist_ok=True)

-    booster.save_model(model, os.path.join(save_dir, "modeling"), shard=True)
+    if use_lora:
+        booster.save_lora_as_pretrained(model, os.path.join(save_dir, "modeling"))
+    else:
+        booster.save_model(model, os.path.join(save_dir, "modeling"), shard=True)

    booster.save_optimizer(optimizer, os.path.join(save_dir, "optimizer"), shard=True)
    booster.save_lr_scheduler(lr_scheduler, os.path.join(save_dir, "lr_scheduler"))