From 6fe3cde820c63d8ba78ce7d01ae000f4fc029517 Mon Sep 17 00:00:00 2001 From: Yang Yu <35400026+YangYu-NUAA@users.noreply.github.com> Date: Mon, 9 Feb 2026 21:15:28 +0800 Subject: [PATCH] =?UTF-8?q?New=20PR=20about=20GLM-4.7-Flash-Lora=E5=BE=AE?= =?UTF-8?q?=E8=B0=83=EF=BC=8C=E5=A2=9E=E5=8A=A0swanlab=E8=AF=A6=E7=BB=86?= =?UTF-8?q?=E8=AF=B4=E6=98=8E=20(#489)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../GLM-4.7-Flash/03-GLM-4.7-Flash-Lora.ipynb | 357 +++++++++++++++--- .../03-GLM-4.7-Flash-Lora微调及Docker镜像.md | 48 +++ 2 files changed, 343 insertions(+), 62 deletions(-) diff --git a/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora.ipynb b/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora.ipynb index 949c987..e7bee7a 100644 --- a/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora.ipynb +++ b/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora.ipynb @@ -90,7 +90,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "2ad7e2de8ae44462b251e81f9ee2e43a", + "model_id": "f7c6adfc9714471b98b4863a583fdcd7", "version_major": 2, "version_minor": 0 }, @@ -108,28 +108,28 @@ "\u001b[1mGlm4MoeLiteForCausalLM LOAD REPORT\u001b[0m from: /root/autodl-fs/ZhipuAI/GLM-4.7-Flash\n", "Key | Status | | \n", "----------------------------------------------------+------------+--+-\n", - "model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n", - "model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n", - "model.layers.47.eh_proj.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n", - "model.layers.47.hnorm.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n", - "model.layers.47.input_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.enorm.weight | UNEXPECTED | | \n", - "model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n", "model.layers.47.mlp.shared_experts.up_proj.weight | UNEXPECTED | | \n", "model.layers.47.shared_head.head.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n", - "model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.gate.weight | UNEXPECTED | | \n", "model.layers.47.post_attention_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n", "model.layers.47.embed_tokens.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n", + "model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n", + "model.layers.47.hnorm.weight | UNEXPECTED | | \n", + "model.layers.47.eh_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n", + "model.layers.47.enorm.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n", + "model.layers.47.input_layernorm.weight | UNEXPECTED | | \n", "model.layers.47.self_attn.kv_a_layernorm.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.gate.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n", + "model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n", "\n", "\u001b[3mNotes:\n", "- UNEXPECTED\u001b[3m\t:can be ignored when loading from different task/architecture; not ok if you expect identical arch.\u001b[0m\n" @@ -206,7 +206,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "bda409422d4e43e399fe79b40479cd23", + "model_id": "48d5005a52cb4ff89bebd65418c88b14", "version_major": 2, "version_minor": 0 }, @@ -224,28 +224,28 @@ "\u001b[1mGlm4MoeLiteForCausalLM LOAD REPORT\u001b[0m from: /root/autodl-fs/ZhipuAI/GLM-4.7-Flash\n", "Key | Status | | \n", "----------------------------------------------------+------------+--+-\n", - "model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n", - "model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n", - "model.layers.47.eh_proj.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n", - "model.layers.47.hnorm.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n", - "model.layers.47.input_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.enorm.weight | UNEXPECTED | | \n", - "model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n", "model.layers.47.mlp.shared_experts.up_proj.weight | UNEXPECTED | | \n", "model.layers.47.shared_head.head.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n", - "model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.gate.weight | UNEXPECTED | | \n", "model.layers.47.post_attention_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n", "model.layers.47.embed_tokens.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n", + "model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n", + "model.layers.47.hnorm.weight | UNEXPECTED | | \n", + "model.layers.47.eh_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n", + "model.layers.47.enorm.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n", + "model.layers.47.input_layernorm.weight | UNEXPECTED | | \n", "model.layers.47.self_attn.kv_a_layernorm.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.gate.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n", + "model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n", "\n", "\u001b[3mNotes:\n", "- UNEXPECTED\u001b[3m\t:can be ignored when loading from different task/architecture; not ok if you expect identical arch.\u001b[0m\n" @@ -283,7 +283,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "4006397e705f4061acb7e447ce6e74fb", + "model_id": "9d2c7ab198834074b0388638707e1442", "version_major": 2, "version_minor": 0 }, @@ -1282,6 +1282,217 @@ "None\n" ] }, + { + "data": { + "application/vnd.jupyter.widget-view+json": { + "model_id": "6589018c8c3a4e12a69ad64ec9d2c18f", + "version_major": 2, + "version_minor": 0 + }, + "text/plain": [ + "Output()" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/html": [ + "
\n"
+      ],
+      "text/plain": []
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "application/vnd.jupyter.widget-view+json": {
+       "model_id": "b705868327e244b8bd0b2b035231ea61",
+       "version_major": 2,
+       "version_minor": 0
+      },
+      "text/plain": [
+       "Output()"
+      ]
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "text/html": [
+       "
\n"
+      ],
+      "text/plain": []
+     },
+     "metadata": {},
+     "output_type": "display_data"
+    },
+    {
+     "data": {
+      "text/html": [
+       "
swanlab: Tracking run with swanlab version 0.7.8\n",
+       "
\n" + ], + "text/plain": [ + "\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m Tracking run with swanlab version \u001b[1;36m0.7\u001b[0m.\u001b[1;36m8\u001b[0m\n" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/html": [ + "
swanlab: Run data will be saved locally in /root/swanlog/run-20260209_010330-xh4gul4jv9luh4leulb3p\n",
+       "
\n" + ], + "text/plain": [ + "\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m Run data will be saved locally in \u001b[1;35m/root/swanlog/run-20260209_010330-xh4gul4jv9luh4leulb3p\u001b[0m\n" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/html": [ + "
swanlab: 👋 Hi yy4192009,welcome to swanlab!\n",
+       "
\n" + ], + "text/plain": [ + "\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m\u001b[1;34m \u001b[0m👋 Hi \u001b[1;39myy4192009\u001b[0m,welcome to swanlab!\n" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/html": [ + "
swanlab: Syncing run ox-3 to the cloud\n",
+       "
\n" + ], + "text/plain": [ + "\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m Syncing run \u001b[33mox-3\u001b[0m to the cloud\n" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/html": [ + "
swanlab: 🏠 View project at https://swanlab.cn/@yy4192009/self-llm\n",
+       "
\n" + ], + "text/plain": [ + "\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m 🏠 View project at \u001b[4;34mhttps://swanlab.cn/@yy4192009/self-llm\u001b[0m\n" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/html": [ + "
swanlab: 🚀 View run at https://swanlab.cn/@yy4192009/self-llm/runs/xh4gul4jv9luh4leulb3p\n",
+       "
\n" + ], + "text/plain": [ + "\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m 🚀 View run at \u001b[4;34mhttps://swanlab.cn/@yy4192009/self-llm/runs/xh4gul4jv9luh4leulb3p\u001b[0m\n" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/html": [ + "\n", + "\n", + "\n", + "\n", + " \n", + " \n", + " Show Iframe\n", + " \n", + " \n", + " \n", + "\n", + "\n", + "

" + ], + "text/plain": [ + "" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, { "data": { "text/html": [ @@ -1289,7 +1500,7 @@ "
\n", " \n", " \n", - " [30/30 03:08, Epoch 1/1]\n", + " [30/30 03:09, Epoch 1/1]\n", "
\n", " \n", " \n", @@ -1301,15 +1512,15 @@ " \n", " \n", " \n", - " \n", + " \n", " \n", " \n", " \n", - " \n", + " \n", " \n", " \n", " \n", - " \n", + " \n", " \n", " \n", "
103.3392063.341679
203.1118203.111885
303.0885053.088572

" @@ -1324,7 +1535,7 @@ { "data": { "text/plain": [ - "TrainOutput(global_step=30, training_loss=3.179843457539876, metrics={'train_runtime': 197.2214, 'train_samples_per_second': 18.908, 'train_steps_per_second': 0.152, 'total_flos': 9.046072556851046e+16, 'train_loss': 3.179843457539876, 'epoch': 1.0})" + "TrainOutput(global_step=30, training_loss=3.1807120005289713, metrics={'train_runtime': 198.5129, 'train_samples_per_second': 18.785, 'train_steps_per_second': 0.151, 'total_flos': 9.046072556851046e+16, 'train_loss': 3.1807120005289713, 'epoch': 1.0})" ] }, "execution_count": 10, @@ -1358,12 +1569,34 @@ " report_to=\"none\",\n", ")\n", "\n", + "import swanlab\n", + "from swanlab.integration.transformers import SwanLabCallback\n", + "\n", + "swanlab.login(api_key='uFd7skQP76EraVKAtbkLZ', save=True) # 记得替换为自己账号的apikey\n", + "\n", + "run = swanlab.init(\n", + " # 设置项目\n", + " project=\"self-llm\",\n", + " # 跟踪超参数与实验元数据\n", + " config={\n", + " \"learning_rate\": 1e-4,\n", + " \"epochs\": 1,\n", + " },\n", + ")\n", + "\n", + "# 实例化SwanLabCallback\n", + "swanlab_callback = SwanLabCallback(\n", + " project=\"self-llm\", \n", + " experiment_name=\"glm4.7-flash-lora\"\n", + ")\n", + "\n", "\n", "trainer = Trainer(\n", " model=model,\n", " args=args,\n", " train_dataset=tokenized_id,\n", " data_collator=DataCollatorForSeq2Seq(tokenizer=tokenizer, padding=True),\n", + " callbacks=[swanlab_callback],\n", ")\n", "\n", "trainer.train()" @@ -1382,14 +1615,14 @@ }, { "cell_type": "code", - "execution_count": 13, + "execution_count": 12, "id": "a6898745-1146-415d-8798-43731d07d47e", "metadata": {}, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "65a2af217265442e816e35cb90e824e3", + "model_id": "a1dd990452814dd1a82790b5eb96913a", "version_major": 2, "version_minor": 0 }, @@ -1407,28 +1640,28 @@ "\u001b[1mGlm4MoeLiteForCausalLM LOAD REPORT\u001b[0m from: /root/autodl-fs/ZhipuAI/GLM-4.7-Flash\n", "Key | Status | | \n", "----------------------------------------------------+------------+--+-\n", - "model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n", - "model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n", - "model.layers.47.eh_proj.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n", - "model.layers.47.hnorm.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n", - "model.layers.47.input_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.enorm.weight | UNEXPECTED | | \n", - "model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n", "model.layers.47.mlp.shared_experts.up_proj.weight | UNEXPECTED | | \n", "model.layers.47.shared_head.head.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n", - "model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.gate.weight | UNEXPECTED | | \n", "model.layers.47.post_attention_layernorm.weight | UNEXPECTED | | \n", - "model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n", "model.layers.47.embed_tokens.weight | UNEXPECTED | | \n", - "model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n", + "model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n", + "model.layers.47.hnorm.weight | UNEXPECTED | | \n", + "model.layers.47.eh_proj.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n", + "model.layers.47.enorm.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n", + "model.layers.47.input_layernorm.weight | UNEXPECTED | | \n", "model.layers.47.self_attn.kv_a_layernorm.weight | UNEXPECTED | | \n", + "model.layers.47.mlp.gate.weight | UNEXPECTED | | \n", + "model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n", + "model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n", "\n", "\u001b[3mNotes:\n", "- UNEXPECTED\u001b[3m\t:can be ignored when loading from different task/architecture; not ok if you expect identical arch.\u001b[0m\n" @@ -1438,7 +1671,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "1. **分析用户输入:**用户问“你是谁?”。这表明他们不知道我是谁,或者正在测试我。我需要介绍自己。我是甄嬛。皇上,您怎么来了?怎么不让人通报一声?您怎么不穿外衣?怎么不穿外衣?怎么不\n" + "1. 甄嬛。皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么\n" ] } ], diff --git a/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora微调及Docker镜像.md b/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora微调及Docker镜像.md index 036ce2d..e88b466 100644 --- a/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora微调及Docker镜像.md +++ b/models/GLM-4.7-Flash/03-GLM-4.7-Flash-Lora微调及Docker镜像.md @@ -120,6 +120,52 @@ text = tokenizer.apply_chat_template( print(text) ``` +## SwanLab简介 + +[SwanLab](https://github.com/swanhubx/swanlab) 是一个开源的模型训练记录工具,面向AI研究者,提供了训练可视化、自动日志记录、超参数记录、实验对比、多人协同等功能。在SwanLab上,研究者能基于直观的可视化图表发现训练问题,对比多个实验找到研究灵感,并通过在线链接的分享与基于组织的多人协同训练,打破团队沟通的壁垒。 + +**为什么要记录训练** + +相较于软件开发,模型训练更像一个实验科学。一个品质优秀的模型背后,往往是成千上万次实验。研究者需要不断尝试、记录、对比,积累经验,才能找到最佳的模型结构、超参数与数据配比。在这之中,如何高效进行记录与对比,对于研究效率的提升至关重要。 + +SwanLab与Transformers已经做好了集成,用法是在Trainer的 `callbacks`参数中添加 `SwanLabCallback`实例,就可以自动记录超参数和训练指标,简化代码如下: + +``` +import swanlab +from swanlab.integration.transformers import SwanLabCallback + +swanlab.login(api_key='your-apikey', save=True) # 记得替换为自己账号的apikey + +run = swanlab.init( + # 设置项目 + project="self-llm", + # 跟踪超参数与实验元数据 + config={ + "learning_rate": 1e-4, + "epochs": 1, + }, +) + +# 实例化SwanLabCallback +swanlab_callback = SwanLabCallback( + project="self-llm", + experiment_name="glm4.7-flash-lora" +) + + +trainer = Trainer( + model=model, + args=args, + train_dataset=tokenized_id, + data_collator=DataCollatorForSeq2Seq(tokenizer=tokenizer, padding=True), + callbacks=[swanlab_callback], +) +``` + +首次使用SwanLab,需要先在[官网](https://swanlab.cn/)注册一个账号,然后在用户设置页面复制你的API Key,然后在训练开始提示登录时粘贴即可,后续无需再次登录。 + +更多用法可参考[快速开始](https://docs.swanlab.cn/zh/guide_cloud/general/quick-start.html)、[Transformers集成](https://docs.swanlab.cn/zh/guide_cloud/integration/integration-huggingface-transformers.html)。 + ## 加载模型和 tokenizer 注意,最好使用 `Glm4MoeLiteForCausalLM`类加载模型 @@ -238,3 +284,5 @@ print(tokenizer.decode(outputs[0][len(inputs[0]):], skip_special_tokens=True)) ``` 甄嬛。臣女家父是太医院院判甄远道。臣女家父与太医院有旧,臣女自幼便在太医院长大 ``` + +