New PR about GLM-4.7-Flash-Lora微调,增加swanlab详细说明 (#489)

This commit is contained in:
Yang Yu
2026-02-09 21:15:28 +08:00
committed by GitHub
parent 8a9d6616bc
commit 6fe3cde820
2 changed files with 343 additions and 62 deletions
+295 -62
View File
@@ -90,7 +90,7 @@
{
"data": {
"application/vnd.jupyter.widget-view+json": {
"model_id": "2ad7e2de8ae44462b251e81f9ee2e43a",
"model_id": "f7c6adfc9714471b98b4863a583fdcd7",
"version_major": 2,
"version_minor": 0
},
@@ -108,28 +108,28 @@
"\u001b[1mGlm4MoeLiteForCausalLM LOAD REPORT\u001b[0m from: /root/autodl-fs/ZhipuAI/GLM-4.7-Flash\n",
"Key | Status | | \n",
"----------------------------------------------------+------------+--+-\n",
"model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n",
"model.layers.47.eh_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.hnorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n",
"model.layers.47.input_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.enorm.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.up_proj.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.head.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.weight | UNEXPECTED | | \n",
"model.layers.47.post_attention_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n",
"model.layers.47.embed_tokens.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n",
"model.layers.47.hnorm.weight | UNEXPECTED | | \n",
"model.layers.47.eh_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n",
"model.layers.47.enorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n",
"model.layers.47.input_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n",
"\n",
"\u001b[3mNotes:\n",
"- UNEXPECTED\u001b[3m\t:can be ignored when loading from different task/architecture; not ok if you expect identical arch.\u001b[0m\n"
@@ -206,7 +206,7 @@
{
"data": {
"application/vnd.jupyter.widget-view+json": {
"model_id": "bda409422d4e43e399fe79b40479cd23",
"model_id": "48d5005a52cb4ff89bebd65418c88b14",
"version_major": 2,
"version_minor": 0
},
@@ -224,28 +224,28 @@
"\u001b[1mGlm4MoeLiteForCausalLM LOAD REPORT\u001b[0m from: /root/autodl-fs/ZhipuAI/GLM-4.7-Flash\n",
"Key | Status | | \n",
"----------------------------------------------------+------------+--+-\n",
"model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n",
"model.layers.47.eh_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.hnorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n",
"model.layers.47.input_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.enorm.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.up_proj.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.head.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.weight | UNEXPECTED | | \n",
"model.layers.47.post_attention_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n",
"model.layers.47.embed_tokens.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n",
"model.layers.47.hnorm.weight | UNEXPECTED | | \n",
"model.layers.47.eh_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n",
"model.layers.47.enorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n",
"model.layers.47.input_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n",
"\n",
"\u001b[3mNotes:\n",
"- UNEXPECTED\u001b[3m\t:can be ignored when loading from different task/architecture; not ok if you expect identical arch.\u001b[0m\n"
@@ -283,7 +283,7 @@
{
"data": {
"application/vnd.jupyter.widget-view+json": {
"model_id": "4006397e705f4061acb7e447ce6e74fb",
"model_id": "9d2c7ab198834074b0388638707e1442",
"version_major": 2,
"version_minor": 0
},
@@ -1282,6 +1282,217 @@
"None\n"
]
},
{
"data": {
"application/vnd.jupyter.widget-view+json": {
"model_id": "6589018c8c3a4e12a69ad64ec9d2c18f",
"version_major": 2,
"version_minor": 0
},
"text/plain": [
"Output()"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"></pre>\n"
],
"text/plain": []
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"application/vnd.jupyter.widget-view+json": {
"model_id": "b705868327e244b8bd0b2b035231ea61",
"version_major": 2,
"version_minor": 0
},
"text/plain": [
"Output()"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"></pre>\n"
],
"text/plain": []
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"color: #000080; text-decoration-color: #000080; font-weight: bold\">swanlab</span><span style=\"color: #000000; text-decoration-color: #000000; font-weight: bold\">:</span> Tracking run with swanlab version <span style=\"color: #008080; text-decoration-color: #008080; font-weight: bold\">0.7</span>.<span style=\"color: #008080; text-decoration-color: #008080; font-weight: bold\">8</span>\n",
"</pre>\n"
],
"text/plain": [
"\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m Tracking run with swanlab version \u001b[1;36m0.7\u001b[0m.\u001b[1;36m8\u001b[0m\n"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"color: #000080; text-decoration-color: #000080; font-weight: bold\">swanlab</span><span style=\"color: #000000; text-decoration-color: #000000; font-weight: bold\">:</span> Run data will be saved locally in <span style=\"color: #800080; text-decoration-color: #800080; font-weight: bold\">/root/swanlog/run-20260209_010330-xh4gul4jv9luh4leulb3p</span>\n",
"</pre>\n"
],
"text/plain": [
"\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m Run data will be saved locally in \u001b[1;35m/root/swanlog/run-20260209_010330-xh4gul4jv9luh4leulb3p\u001b[0m\n"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"color: #000080; text-decoration-color: #000080; font-weight: bold\">swanlab</span><span style=\"color: #000000; text-decoration-color: #000000; font-weight: bold\">:</span><span style=\"color: #000080; text-decoration-color: #000080; font-weight: bold\"> </span>👋 Hi <span style=\"color: #000000; text-decoration-color: #000000; font-weight: bold\">yy4192009</span>,welcome to swanlab!\n",
"</pre>\n"
],
"text/plain": [
"\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m\u001b[1;34m \u001b[0m👋 Hi \u001b[1;39myy4192009\u001b[0m,welcome to swanlab!\n"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"color: #000080; text-decoration-color: #000080; font-weight: bold\">swanlab</span><span style=\"color: #000000; text-decoration-color: #000000; font-weight: bold\">:</span> Syncing run <span style=\"color: #808000; text-decoration-color: #808000\">ox-3</span> to the cloud\n",
"</pre>\n"
],
"text/plain": [
"\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m Syncing run \u001b[33mox-3\u001b[0m to the cloud\n"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"color: #000080; text-decoration-color: #000080; font-weight: bold\">swanlab</span><span style=\"color: #000000; text-decoration-color: #000000; font-weight: bold\">:</span> 🏠 View project at <span style=\"color: #000080; text-decoration-color: #000080; text-decoration: underline\">https://swanlab.cn/@yy4192009/self-llm</span>\n",
"</pre>\n"
],
"text/plain": [
"\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m 🏠 View project at \u001b[4;34mhttps://swanlab.cn/@yy4192009/self-llm\u001b[0m\n"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"<pre style=\"white-space:pre;overflow-x:auto;line-height:normal;font-family:Menlo,'DejaVu Sans Mono',consolas,'Courier New',monospace\"><span style=\"color: #000080; text-decoration-color: #000080; font-weight: bold\">swanlab</span><span style=\"color: #000000; text-decoration-color: #000000; font-weight: bold\">:</span> 🚀 View run at <span style=\"color: #000080; text-decoration-color: #000080; text-decoration: underline\">https://swanlab.cn/@yy4192009/self-llm/runs/xh4gul4jv9luh4leulb3p</span>\n",
"</pre>\n"
],
"text/plain": [
"\u001b[1;34mswanlab\u001b[0m\u001b[1;39m:\u001b[0m 🚀 View run at \u001b[4;34mhttps://swanlab.cn/@yy4192009/self-llm/runs/xh4gul4jv9luh4leulb3p\u001b[0m\n"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
"\n",
"<!DOCTYPE html>\n",
"<html lang=\"en\">\n",
"<head>\n",
" <meta charset=\"UTF-8\">\n",
" <meta name=\"viewport\" content=\"width=device-width, initial-scale=1.0\">\n",
" <title>Show Iframe</title>\n",
" \n",
" <script>\n",
" function showIframe() {\n",
" var iframeHtml = '<iframe src=\"https://swanlab.cn/@yy4192009/self-llm/runs/xh4gul4jv9luh4leulb3p\" width=100% height=\"600\" frameborder=\"no\"></iframe>';\n",
" document.getElementById('iframeContainer').innerHTML = iframeHtml;\n",
" }\n",
" </script>\n",
" \n",
"</head>\n",
"<body>\n",
" <style>\n",
" .interactive-button {\n",
" display: flex;\n",
" align-items: center;\n",
" height: 36px;\n",
" border: 0px;\n",
" background-color: #2c8f63;\n",
" color: white;\n",
" padding: 10px 20px;\n",
" transition: background-color 0.3s, transform 0.2s;\n",
" }\n",
"\n",
" .interactive-button:hover {\n",
" background-color: #5cab87;\n",
" cursor: pointer;\n",
" }\n",
"\n",
" .interactive-button:active { background-color: #217952; transform: scale(0.96); } </style> <br> <button \n",
" onclick=\"showIframe()\" class=\"interactive-button\"> <svg style=\"height: 16px; margin-right: 8px;\" viewBox=\"0 0 \n",
" 46 46\" fill=\"none\"> <path d=\"M10.8439 21.1974C10.6414 21.2854 10.4477 21.3925 10.2655 21.5173L10.2069 \n",
" 21.5652C10.1839 21.58 10.1625 21.5969 10.1429 21.6159C6.29135 24.6118 4.22831 29.4416 5.32646 34.282C5.94656 \n",
" 37.0577 7.50461 39.5348 9.73801 41.2958C11.9714 43.0568 14.7436 43.994 17.5874 43.9495H18.0219C19.8864 \n",
" 43.8697 21.7087 43.3694 23.3526 42.486C24.9964 41.6026 26.4193 40.3589 27.5147 38.848C28.61 37.3371 29.3496 \n",
" 35.598 29.678 33.761C30.0065 31.9239 29.9153 30.0363 29.4112 28.2395C28.9181 26.4723 27.8919 24.8437 26.9937 \n",
" 23.2551C25.4158 20.4653 23.8343 17.6764 22.2492 14.8884C21.7801 14.0647 21.3057 13.2465 20.8419 \n",
" 12.4228C20.2315 11.3353 19.2746 10.1519 19.224 8.86183C19.1733 7.57176 20.2235 6.32701 21.5082 \n",
" 6.07912C23.9284 5.61801 25.0639 8.24078 25.0693 8.23812C25.363 8.94035 25.9123 9.50489 26.6063 \n",
" 9.81764C27.3002 10.1304 28.087 10.168 28.8077 9.92298C29.5283 9.67791 30.1291 9.1684 30.4885 8.49743C30.8479 \n",
" 7.82646 30.9392 7.04405 30.7439 6.30835C30.1514 4.37314 28.9133 2.69953 27.2363 1.56656C25.7615 0.511704 \n",
" 23.9847 -0.0372109 22.1719 0.00195984C20.9049 0.00893199 19.6532 0.27989 18.4967 0.797557C17.3402 1.31522 \n",
" 16.3043 2.06823 15.4551 3.00856C14.49 4.08707 13.7984 5.38193 13.4389 6.78385C13.0794 8.18576 13.0624 9.6536 \n",
" 13.3894 11.0635C13.52 11.593 13.6984 12.1095 13.9225 12.6067C14.5595 14.0514 15.4951 15.3681 16.284 \n",
" 16.7355C17.2525 18.4147 18.2209 20.0948 19.1893 21.7758C20.1578 23.4568 21.1351 25.1449 22.1213 \n",
" 26.8401C22.9209 28.2421 23.7925 29.4682 23.8805 31.1528C23.9175 32.0513 23.7682 32.9479 23.4419 \n",
" 33.7859C23.1156 34.6239 22.6194 35.3854 21.9845 36.0223C21.3496 36.6592 20.5897 37.1578 19.7527 \n",
" 37.4868C18.9157 37.8157 18.0196 37.9678 17.121 37.9336C14.0024 37.7923 11.6488 35.4814 11.1744 32.4588C10.58 \n",
" 28.6419 13.552 26.5469 13.552 26.5469C14.1782 26.1785 14.6497 25.5955 14.8791 24.906C15.1084 24.2166 15.0801 \n",
" 23.4673 14.7993 22.7971C14.5186 22.127 14.0044 21.5813 13.3521 21.2611C12.6998 20.941 11.9536 20.8682 11.2517 \n",
" 21.0561C11.1174 21.0939 10.9856 21.1402 10.8572 21.1947\" fill=\"white\" /> <path d=\"M42.8101 31.5968C42.8109 \n",
" 30.5198 42.7218 29.4445 42.5435 28.3823C42.2663 26.7069 41.7464 25.0808 41.0002 23.5552C40.5524 22.6463 \n",
" 39.9874 21.7374 39.1024 21.2417C38.6593 20.9919 38.1589 20.8617 37.6502 20.8639C37.1416 20.8661 36.6423 \n",
" 21.0006 36.2013 21.2541C35.7604 21.5077 35.393 21.8716 35.1352 22.3101C34.8775 22.7485 34.7382 23.2466 \n",
" 34.7312 23.7552C34.7072 24.8773 35.3149 25.8875 35.768 26.9217C36.5212 28.6453 36.8623 30.5208 36.7642 \n",
" 32.3993C36.6661 34.2777 36.1315 36.1075 35.2029 37.7433C35.146 37.8404 35.0952 37.941 35.051 38.0445C34.8623 \n",
" 38.4842 34.7635 38.9573 34.7605 39.4358C34.7802 40.1222 35.0356 40.7808 35.4835 41.3011C35.9315 41.8214 \n",
" 36.5449 42.1717 37.2207 42.2932C38.8759 42.589 40.1899 41.347 40.8856 39.9609C42.1643 37.3589 42.823 34.4961 \n",
" 42.8101 31.5968Z\" fill=\"white\" /> <path d=\"M28.2309 11.8938C28.1761 11.9043 28.1218 11.9176 28.0683 \n",
" 11.9338C27.9593 11.9642 27.8611 12.0249 27.7851 12.1088C27.7091 12.1928 27.6584 12.2965 27.6389 \n",
" 12.408C27.6193 12.5195 27.6318 12.6343 27.6748 12.7391C27.7178 12.8438 27.7895 12.9343 27.8818 \n",
" 12.9999C29.2375 14.0252 30.3809 15.3043 31.2482 16.7662C31.4838 17.1677 31.6888 17.5865 31.8612 \n",
" 18.0189C32.0052 18.3921 32.1971 18.8799 32.6822 18.8532C33.0607 18.8346 33.2153 18.512 33.3192 \n",
" 18.1895C33.8137 16.5125 33.9678 14.7534 33.7723 13.0159C33.6331 12.0693 33.4155 11.1359 33.122 \n",
" 10.2252C33.0775 10.0047 32.9744 9.80029 32.8235 9.6335C32.7273 9.54627 32.6054 9.49262 32.4761 9.4806C32.3468 \n",
" 9.46859 32.2171 9.49886 32.1065 9.56687C32.0016 9.65188 31.9115 9.75365 31.8399 9.86806C31.3956 10.4658 \n",
" 30.825 10.9581 30.1687 11.3101C29.8377 11.4861 29.4893 11.6272 29.1292 11.7312C28.828 11.8192 28.5215 11.8325 \n",
" 28.2309 11.8938Z\" fill=\"white\" /> </svg> Display SwanLab Board </button> <br> <div \n",
" id=\"iframeContainer\"></div> </body> </html>"
],
"text/plain": [
"<IPython.core.display.HTML object>"
]
},
"metadata": {},
"output_type": "display_data"
},
{
"data": {
"text/html": [
@@ -1289,7 +1500,7 @@
" <div>\n",
" \n",
" <progress value='30' max='30' style='width:300px; height:20px; vertical-align: middle;'></progress>\n",
" [30/30 03:08, Epoch 1/1]\n",
" [30/30 03:09, Epoch 1/1]\n",
" </div>\n",
" <table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
@@ -1301,15 +1512,15 @@
" <tbody>\n",
" <tr>\n",
" <td>10</td>\n",
" <td>3.339206</td>\n",
" <td>3.341679</td>\n",
" </tr>\n",
" <tr>\n",
" <td>20</td>\n",
" <td>3.111820</td>\n",
" <td>3.111885</td>\n",
" </tr>\n",
" <tr>\n",
" <td>30</td>\n",
" <td>3.088505</td>\n",
" <td>3.088572</td>\n",
" </tr>\n",
" </tbody>\n",
"</table><p>"
@@ -1324,7 +1535,7 @@
{
"data": {
"text/plain": [
"TrainOutput(global_step=30, training_loss=3.179843457539876, metrics={'train_runtime': 197.2214, 'train_samples_per_second': 18.908, 'train_steps_per_second': 0.152, 'total_flos': 9.046072556851046e+16, 'train_loss': 3.179843457539876, 'epoch': 1.0})"
"TrainOutput(global_step=30, training_loss=3.1807120005289713, metrics={'train_runtime': 198.5129, 'train_samples_per_second': 18.785, 'train_steps_per_second': 0.151, 'total_flos': 9.046072556851046e+16, 'train_loss': 3.1807120005289713, 'epoch': 1.0})"
]
},
"execution_count": 10,
@@ -1358,12 +1569,34 @@
" report_to=\"none\",\n",
")\n",
"\n",
"import swanlab\n",
"from swanlab.integration.transformers import SwanLabCallback\n",
"\n",
"swanlab.login(api_key='uFd7skQP76EraVKAtbkLZ', save=True) # 记得替换为自己账号的apikey\n",
"\n",
"run = swanlab.init(\n",
" # 设置项目\n",
" project=\"self-llm\",\n",
" # 跟踪超参数与实验元数据\n",
" config={\n",
" \"learning_rate\": 1e-4,\n",
" \"epochs\": 1,\n",
" },\n",
")\n",
"\n",
"# 实例化SwanLabCallback\n",
"swanlab_callback = SwanLabCallback(\n",
" project=\"self-llm\", \n",
" experiment_name=\"glm4.7-flash-lora\"\n",
")\n",
"\n",
"\n",
"trainer = Trainer(\n",
" model=model,\n",
" args=args,\n",
" train_dataset=tokenized_id,\n",
" data_collator=DataCollatorForSeq2Seq(tokenizer=tokenizer, padding=True),\n",
" callbacks=[swanlab_callback],\n",
")\n",
"\n",
"trainer.train()"
@@ -1382,14 +1615,14 @@
},
{
"cell_type": "code",
"execution_count": 13,
"execution_count": 12,
"id": "a6898745-1146-415d-8798-43731d07d47e",
"metadata": {},
"outputs": [
{
"data": {
"application/vnd.jupyter.widget-view+json": {
"model_id": "65a2af217265442e816e35cb90e824e3",
"model_id": "a1dd990452814dd1a82790b5eb96913a",
"version_major": 2,
"version_minor": 0
},
@@ -1407,28 +1640,28 @@
"\u001b[1mGlm4MoeLiteForCausalLM LOAD REPORT\u001b[0m from: /root/autodl-fs/ZhipuAI/GLM-4.7-Flash\n",
"Key | Status | | \n",
"----------------------------------------------------+------------+--+-\n",
"model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n",
"model.layers.47.eh_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.hnorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n",
"model.layers.47.input_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.enorm.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.up_proj.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.head.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.weight | UNEXPECTED | | \n",
"model.layers.47.post_attention_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n",
"model.layers.47.embed_tokens.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.e_score_correction_bias | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.down_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.q_a_proj.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_b_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.gate_up_proj | UNEXPECTED | | \n",
"model.layers.47.mlp.shared_experts.gate_proj.weight | UNEXPECTED | | \n",
"model.layers.47.hnorm.weight | UNEXPECTED | | \n",
"model.layers.47.eh_proj.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.experts.down_proj | UNEXPECTED | | \n",
"model.layers.47.enorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_proj_with_mqa.weight | UNEXPECTED | | \n",
"model.layers.47.input_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.kv_a_layernorm.weight | UNEXPECTED | | \n",
"model.layers.47.mlp.gate.weight | UNEXPECTED | | \n",
"model.layers.47.self_attn.o_proj.weight | UNEXPECTED | | \n",
"model.layers.47.shared_head.norm.weight | UNEXPECTED | | \n",
"\n",
"\u001b[3mNotes:\n",
"- UNEXPECTED\u001b[3m\t:can be ignored when loading from different task/architecture; not ok if you expect identical arch.\u001b[0m\n"
@@ -1438,7 +1671,7 @@
"name": "stdout",
"output_type": "stream",
"text": [
"1. **分析用户输入:**用户问“你是谁?”。这表明他们不知道我是谁,或者正在测试我。我需要介绍自己。</think>我是甄嬛。皇上,您怎么来了?怎么不让人通报一声?您怎么不穿外衣?怎么不穿外衣?怎么不\n"
"1. 甄嬛。皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么来了?皇上,您怎么\n"
]
}
],
@@ -120,6 +120,52 @@ text = tokenizer.apply_chat_template(
print(text)
```
## SwanLab简介
[SwanLab](https://github.com/swanhubx/swanlab) 是一个开源的模型训练记录工具,面向AI研究者,提供了训练可视化、自动日志记录、超参数记录、实验对比、多人协同等功能。在SwanLab上,研究者能基于直观的可视化图表发现训练问题,对比多个实验找到研究灵感,并通过在线链接的分享与基于组织的多人协同训练,打破团队沟通的壁垒。
**为什么要记录训练**
相较于软件开发,模型训练更像一个实验科学。一个品质优秀的模型背后,往往是成千上万次实验。研究者需要不断尝试、记录、对比,积累经验,才能找到最佳的模型结构、超参数与数据配比。在这之中,如何高效进行记录与对比,对于研究效率的提升至关重要。
SwanLab与Transformers已经做好了集成,用法是在Trainer的 `callbacks`参数中添加 `SwanLabCallback`实例,就可以自动记录超参数和训练指标,简化代码如下:
```
import swanlab
from swanlab.integration.transformers import SwanLabCallback
swanlab.login(api_key='your-apikey', save=True) # 记得替换为自己账号的apikey
run = swanlab.init(
# 设置项目
project="self-llm",
# 跟踪超参数与实验元数据
config={
"learning_rate": 1e-4,
"epochs": 1,
},
)
# 实例化SwanLabCallback
swanlab_callback = SwanLabCallback(
project="self-llm",
experiment_name="glm4.7-flash-lora"
)
trainer = Trainer(
model=model,
args=args,
train_dataset=tokenized_id,
data_collator=DataCollatorForSeq2Seq(tokenizer=tokenizer, padding=True),
callbacks=[swanlab_callback],
)
```
首次使用SwanLab,需要先在[官网](https://swanlab.cn/)注册一个账号,然后在用户设置页面复制你的API Key,然后在训练开始提示登录时粘贴即可,后续无需再次登录。
更多用法可参考[快速开始](https://docs.swanlab.cn/zh/guide_cloud/general/quick-start.html)、[Transformers集成](https://docs.swanlab.cn/zh/guide_cloud/integration/integration-huggingface-transformers.html)。
## 加载模型和 tokenizer
注意,最好使用 `Glm4MoeLiteForCausalLM`类加载模型
@@ -238,3 +284,5 @@ print(tokenizer.decode(outputs[0][len(inputs[0]):], skip_special_tokens=True))
```
甄嬛。臣女家父是太医院院判甄远道。臣女家父与太医院有旧,臣女自幼便在太医院长大
```