From 5877a4145103b8a4c57a54b0dac6bf019537aa08 Mon Sep 17 00:00:00 2001 From: KMnO4-zx <1021385881@qq.com> Date: Sun, 21 Jul 2024 22:05:48 +0800 Subject: [PATCH] =?UTF-8?q?=E4=BC=98=E5=8C=96=E9=A1=B9=E7=9B=AE=E7=BB=93?= =?UTF-8?q?=E6=9E=84=EF=BC=8C=E5=A2=9E=E5=8A=A0=20Examples=20ToDo?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- README.md | 256 +++--- examples/readme.md | 13 + .../Atom}/01-Atom-7B-chat-WebDemo.md | 0 .../Atom}/02-Atom-7B-Chat Lora 微调.md | 0 .../Atom}/02-Atom-7B-Chat-Lora/train.py | 0 .../Atom}/02-Atom-7B-Chat-Lora/train.sh | 0 ...tom-7B-Chat 接入langchain搭建知识库助手.md | 0 .../LLM.py | 0 .../creat_db.py | 0 .../readme.md | 0 .../run_gradio.py | 0 .../Atom}/04-Atom-7B-chat 全量微调.md | 0 {Atom => models/Atom}/images/image-1.png | Bin {Atom => models/Atom}/images/image-2.png | Bin {Atom => models/Atom}/images/image-3.png | Bin {Atom => models/Atom}/images/image-4.png | Bin {Atom => models/Atom}/images/image-5.png | Bin {Atom => models/Atom}/images/image-6.png | Bin {Atom => models/Atom}/images/image-7.png | Bin {Atom => models/Atom}/images/image-8.png | Bin {Atom => models/Atom}/images/image-9.png | Bin .../01-Baichuan2-7B-chat+FastApi+部署调用.md | 0 .../BaiChuan}/02-Baichuan-7B-chat+WebDemo.md | 0 .../03-Baichuan2-7B-chat接入LangChain框架.md | 0 .../04-Baichuan2-7B-chat Lora 微调.ipynb | 0 .../04-Baichuan2-7B-chat+lora+微调.md | 0 .../BaiChuan}/images/image1.png | Bin .../BaiChuan}/images/image10.png | Bin .../BaiChuan}/images/image11.png | Bin .../BaiChuan}/images/image12.png | Bin .../BaiChuan}/images/image13.png | Bin .../BaiChuan}/images/image14.png | Bin .../BaiChuan}/images/image15.png | Bin .../BaiChuan}/images/image16.png | Bin .../BaiChuan}/images/image17.png | Bin .../BaiChuan}/images/image18.png | Bin .../BaiChuan}/images/image2.png | Bin .../BaiChuan}/images/image20.png | Bin .../BaiChuan}/images/image23.png | Bin .../BaiChuan}/images/image25.png | Bin .../BaiChuan}/images/image26.png | Bin .../BaiChuan}/images/image27.png | Bin .../BaiChuan}/images/image3.png | Bin .../BaiChuan}/images/image4.png | Bin .../BaiChuan}/images/image6.png | Bin .../BaiChuan}/images/image7.png | Bin .../BaiChuan}/images/image8.png | Bin .../BaiChuan}/images/image9.png | Bin .../BlueLM}/01-BlueLM-7B-Chat FastApi 部署.md | 0 .../02-BlueLM-7B-Chat langchain 接入.md | 0 .../BlueLM}/03-BlueLM-7B-Chat WebDemo 部署.md | 0 .../BlueLM}/04-BlueLM-7B-Chat Lora 微调.ipynb | 0 .../BlueLM}/04-BlueLM-7B-Chat Lora 微调.md | 0 .../BlueLM}/04-BlueLM-7B-Chat Lora 微调.py | 0 .../BlueLM}/images/202403191628941.png | Bin .../BlueLM}/images/202403191813385.png | Bin .../BlueLM}/images/202403201210690.png | Bin .../BlueLM}/images/202403201229542.png | Bin .../BlueLM}/images/202403202153465.png | Bin .../01-CharacterGLM-6B Transformer部署调用.md | 154 ++-- .../02-CharacterGLM-6B FastApi部署调用.md | 366 ++++----- .../CharacterGLM}/03-CharacterGLM-6B-chat.md | 192 ++--- .../04-CharacterGLM-6B Lora微调.md | 390 ++++----- .../04-CharacterGLM-6B-Lora微调.ipynb | 0 .../04-CharacterGLM-6B-Lora微调.py | 0 .../CharacterGLM}/image/03-webdemo_show.png | Bin .../CharacterGLM}/image/03-修改路径.png | Bin .../CharacterGLM}/image/03-运行clidemo.png | Bin .../CharacterGLM}/image/03-运行webdemo.png | Bin .../CharacterGLM}/image/image-1.png | Bin .../CharacterGLM}/image/image-2.png | Bin .../CharacterGLM}/image/image-3.png | Bin .../CharacterGLM}/image/image-4.png | Bin .../CharacterGLM}/image/readme.md | 0 .../CharacterGLM}/readme.md | 0 .../01-ChatGLM3-6B Transformer部署调用.md | 0 .../02-ChatGLM3-6B FastApi部署调用.md | 0 .../ChatGLM}/03-ChatGLM3-6B-chat.md | 0 .../04-ChatGLM3-6B-Code-Interpreter.md | 0 ...-ChatGLM3-6B接入LangChain搭建知识库助手.md | 0 .../LLM.py | 0 .../create_db.py | 0 .../run_gradio.py | 0 .../ChatGLM}/06-ChatGLM3-6B-Lora微调.ipynb | 0 .../ChatGLM}/06-ChatGLM3-6B-Lora微调.md | 0 .../ChatGLM}/06-ChatGLM3-6B-Lora微调.py | 0 .../ChatGLM}/images/image-1.png | Bin .../ChatGLM}/images/image-2.png | Bin .../ChatGLM}/images/image-3.png | Bin .../ChatGLM}/images/image-4.png | Bin .../ChatGLM}/images/image-5.png | Bin .../ChatGLM}/images/image-6.png | Bin .../ChatGLM}/images/image-7.png | Bin .../ChatGLM}/images/image-8.png | Bin .../ChatGLM}/images/image-9.png | Bin ...Coder-V2-Lite-Instruct FastApi 部署调用.md | 0 ...k-Coder-V2-Lite-Instruct 接入 LangChain.md | 0 ...eek-Coder-V2-Lite-Instruct WebDemo 部署.md | 0 ...eek-Coder-V2-Lite-Instruct Lora 微调.ipynb | 0 ...epSeek-Coder-V2-Lite-Instruct Lora 微调.md | 0 .../DeepSeek-Coder-V2}/images/fig1-1.png | Bin .../DeepSeek-Coder-V2}/images/fig1-2.png | Bin .../DeepSeek-Coder-V2}/images/fig1-3.png | Bin .../DeepSeek-Coder-V2}/images/fig1-4.png | Bin .../DeepSeek-Coder-V2}/images/fig1-5.png | Bin .../DeepSeek-Coder-V2}/images/fig1-6.png | Bin .../DeepSeek-Coder-V2}/images/fig1-7.png | Bin .../DeepSeek-Coder-V2}/images/fig1-8.png | Bin .../DeepSeek-Coder-V2}/images/fig1-9.png | Bin .../DeepSeek-Coder-V2}/images/fig2-1.png | Bin .../DeepSeek-Coder-V2}/images/fig2-2.png | Bin .../DeepSeek-Coder-V2}/images/fig2-3.png | Bin .../DeepSeek-Coder-V2}/images/image03-1.png | Bin .../DeepSeek-Coder-V2}/images/image03-2.png | Bin .../DeepSeek-Coder-V2}/images/image03-3.png | Bin .../DeepSeek-Coder-V2}/images/image03-4.png | Bin .../DeepSeek-Coder-V2}/images/image03-5.png | Bin .../DeepSeek-Coder-V2}/images/image03-6.png | Bin .../DeepSeek}/01-DeepSeek-7B-chat FastApi.md | 0 .../02-DeepSeek-7B-chat langchain.md | 0 .../DeepSeek}/03-DeepSeek-7B-chat WebDemo.md | 0 .../04-DeepSeek-7B-chat Lora 微调.ipynb | 0 .../04-DeepSeek-7B-chat Lora 微调.md | 0 ...eepSeek-7B-chat 4bits量化 Qlora 微调.ipynb | 0 ...5-DeepSeek-7B-chat 4bits量化 Qlora 微调.md | 0 ...6-DeepSeek-MoE-16b-chat FastApi部署调用.md | 0 ...epSeek-MoE-16b-chat Transformer部署调用.md | 0 .../DeepSeek}/07-deepseek_fine_tune.ipynb | 0 .../DeepSeek}/08-deepseek_web_demo.ipynb | 0 .../DeepSeek}/images/image-1.png | Bin .../DeepSeek}/images/image-2.png | Bin .../DeepSeek}/images/image-3.png | Bin .../DeepSeek}/images/image-4.png | Bin .../DeepSeek}/images/image-5.png | Bin .../DeepSeek}/images/image-6.png | Bin .../DeepSeek}/images/image-7.png | Bin .../DeepSeek}/images/image-8.png | Bin .../DeepSeek}/images/image-9.png | Bin .../01-GLM-4-9B-chat FastApi 部署调用.md | 0 .../GLM-4}/02-GLM-4-9B-chat langchain 接入.md | 0 .../GLM-4}/03-GLM-4-9B-Chat WebDemo.md | 0 .../GLM-4}/04-GLM-4-9B-Chat vLLM 部署调用.md | 0 .../GLM-4}/05-GLM-4-9B-chat Lora 微调.ipynb | 0 .../GLM-4}/05-GLM-4-9B-chat Lora 微调.md | 0 .../GLM-4}/benchmark_throughput.py | 0 {GLM-4 => models/GLM-4}/images/image-1.png | Bin {GLM-4 => models/GLM-4}/images/image01-1.png | Bin {GLM-4 => models/GLM-4}/images/image01-2.png | Bin {GLM-4 => models/GLM-4}/images/image01-3.png | Bin {GLM-4 => models/GLM-4}/images/image01-4.png | Bin {GLM-4 => models/GLM-4}/images/image01-5.png | Bin {GLM-4 => models/GLM-4}/images/image02-1.png | Bin {GLM-4 => models/GLM-4}/images/image03-1.png | Bin {GLM-4 => models/GLM-4}/images/image03-2.png | Bin {GLM-4 => models/GLM-4}/images/image04-1.png | Bin .../01-Gemma-2B-Instruct FastApi 部署调用.md | 0 .../02-Gemma-2B-Instruct langchain 接入.md | 0 .../03-Gemma-2B-Instruct WebDemo 部署.md | 0 .../Gemma}/04-Gemma-2B-Instruct Lora微调.md | 0 .../Gemma}/04-Gemma-2B-Lora微调.ipynb | 0 {Gemma => models/Gemma}/images/image-1.png | Bin {Gemma => models/Gemma}/images/image-2.png | Bin {Gemma => models/Gemma}/images/image-3.png | Bin {Gemma => models/Gemma}/images/image-4.png | Bin {Gemma => models/Gemma}/images/image-5.png | Bin .../01-Gemma-2-9b-it FastApi 部署调用.md | 0 .../02-Gemma-2-9b-it langchain 接入.md | 0 .../Gemma2}/03-Gemma-2-9b-it WebDemo 部署.md | 0 .../04-Gemma-2-9b-it peft lora微调.ipynb | 0 .../Gemma2}/04-Gemma-2-9b-it peft lora微调.md | 0 {Gemma2 => models/Gemma2}/images/01-1.png | Bin {Gemma2 => models/Gemma2}/images/01-4-0.png | Bin {Gemma2 => models/Gemma2}/images/01-4-1.png | Bin {Gemma2 => models/Gemma2}/images/01-5.png | Bin {Gemma2 => models/Gemma2}/images/01-6.png | Bin {Gemma2 => models/Gemma2}/images/01-7.png | Bin {Gemma2 => models/Gemma2}/images/02-1.png | Bin {Gemma2 => models/Gemma2}/images/03-0.png | Bin {Gemma2 => models/Gemma2}/images/03-1.png | Bin {Gemma2 => models/Gemma2}/images/03-2.png | Bin {Gemma2 => models/Gemma2}/images/03-3.png | Bin {Gemma2 => models/Gemma2}/images/04-1.png | Bin {Gemma2 => models/Gemma2}/images/04-2.png | Bin .../General-Setting}/01-pip、conda换源.md | 0 .../General-Setting}/02-AutoDL开放端口.md | 0 .../General-Setting}/03-模型下载.md | 0 .../General-Setting}/04-Issue&PR&update.md | 0 .../General-Setting}/pic/Issue1.png | Bin .../General-Setting}/pic/Issue2.png | Bin .../General-Setting}/pic/PR.png | Bin .../General-Setting}/pic/PR1.png | Bin .../General-Setting}/pic/PR3.png | Bin .../General-Setting}/pic/PR4.png | Bin .../General-Setting}/pic/PR5.png | Bin .../General-Setting}/pic/PR6.png | Bin .../General-Setting}/pic/端口映射.png | Bin ...-InternLM-Chat-7B Transformers 部署调用.md | 0 .../InternLM}/02-internLM-Chat-7B FastApi.md | 0 .../InternLM}/03-InternLM-Chat-7B.md | 0 .../04-Lagent+InternLM-Chat-7B-V1.1.md | 0 .../InternLM}/05-浦语灵笔图文理解&创作.md | 0 .../06-InternLM接入LangChain搭建知识库助手.md | 0 .../LLM.py | 0 .../creat_db.py | 0 .../readme.md | 0 .../run_gradio.py | 0 .../InternLM}/images/image-1.png | Bin .../InternLM}/images/image-10.png | Bin .../InternLM}/images/image-11.png | Bin .../InternLM}/images/image-12.png | Bin .../InternLM}/images/image-13.png | Bin .../InternLM}/images/image-14.png | Bin .../InternLM}/images/image-2.png | Bin .../InternLM}/images/image-3.png | Bin .../InternLM}/images/image-4.png | Bin .../InternLM}/images/image-5.png | Bin .../InternLM}/images/image-6.png | Bin .../InternLM}/images/image-7.png | Bin .../InternLM}/images/image-8.png | Bin .../InternLM}/images/image-9.png | Bin .../InternLM}/images/image.png | Bin .../01-InternLM2-7B-chat FastAPI部署.md | 0 .../02-InternLM2-7B-chat langchain 接入.md | 0 .../03-InternLM2-7B-chat WebDemo 部署.md | 260 +++--- .../04-InternLM2-7B-chat Xtuner Qlora 微调.md | 620 +++++++------- .../dataset/心理大模型-职场焦虑语料.xlsx | Bin {InternLM2 => models/InternLM2}/images/1.png | Bin {InternLM2 => models/InternLM2}/images/2.png | Bin .../InternLM2}/images/3-1.png | Bin .../InternLM2}/images/3-2.png | Bin .../InternLM2}/images/3-3.png | Bin .../InternLM2}/images/3-4.png | Bin .../InternLM2}/images/3-5.png | Bin .../InternLM2}/images/3-6.png | Bin .../InternLM2}/images/3-7.png | Bin .../InternLM2}/images/3-8.png | Bin {InternLM2 => models/InternLM2}/images/3.png | Bin .../InternLM2}/images/4-1.png | Bin .../InternLM2}/images/4-2.png | Bin .../InternLM2}/images/4-3.png | Bin .../InternLM2}/images/4-4.png | Bin .../01-LLaMA3-8B-Instruct FastApi 部署调用.md | 0 .../02-LLaMA3-8B-Instruct langchain 接入.md | 0 .../03-LLaMA3-8B-Instruct WebDemo 部署.md | 0 .../04-LLaMA3-8B-Instruct Lora 微调.md | 0 .../LLaMA3}/LLaMA3-8B-Instruct Lora.ipynb | 0 {LLaMA3 => models/LLaMA3}/images/api_resp.png | Bin .../LLaMA3}/images/api_start.png | Bin {LLaMA3 => models/LLaMA3}/images/image-1.png | Bin {LLaMA3 => models/LLaMA3}/images/image-2.png | Bin {LLaMA3 => models/LLaMA3}/images/image-3.png | Bin .../MiniCPM-2B-chat FastApi 部署调用.md | 0 .../MiniCPM-2B-chat Lora && Full 微调.md | 0 .../MiniCPM}/MiniCPM-2B-chat WebDemo部署.md | 0 .../MiniCPM}/MiniCPM-2B-chat langchain接入.md | 0 .../MiniCPM-2B-chat transformers 部署调用.md | 0 {MiniCPM => models/MiniCPM}/ds_config.json | 0 .../MiniCPM}/images/image-1.png | Bin .../MiniCPM}/images/image-10.png | Bin .../MiniCPM}/images/image-2.png | Bin .../MiniCPM}/images/image-3.png | Bin .../MiniCPM}/images/image-4.png | Bin .../MiniCPM}/images/image-5.png | Bin .../MiniCPM}/images/image-6.png | Bin .../MiniCPM}/images/image-7.png | Bin .../MiniCPM}/images/image-8.png | Bin .../MiniCPM}/images/image-9.png | Bin {MiniCPM => models/MiniCPM}/train.py | 0 {MiniCPM => models/MiniCPM}/train.sh | 0 .../Qwen-Audio}/01-Qwen-Audio-chat FastApi.md | 0 .../Qwen-Audio}/02-Qwen-Audio-chat WebDemo.md | 0 .../Qwen-Audio}/images/image-1.png | Bin .../Qwen-Audio}/images/image-2.png | Bin .../Qwen-Audio}/images/image-3.png | Bin .../Qwen-Audio}/images/image-4.png | Bin .../01-Qwen-7B-Chat Transformers部署调用.md | 0 .../Qwen}/02-Qwen-7B-Chat FastApi 部署调用.md | 0 .../Qwen}/03-Qwen-7B-Chat WebDemo.md | 0 .../Qwen}/04-Qwen-7B-Chat Lora 微调.ipynb | 0 .../Qwen}/04-Qwen-7B-Chat Lora 微调.md | 0 .../Qwen}/04-Qwen-7B-Chat Lora 微调.py | 0 .../Qwen}/05-Qwen-7B-Chat Ptuning 微调.md | 0 .../Qwen}/05-Qwen-7B-Chat Ptuning 微调.py | 0 .../Qwen}/06-Qwen-7B-chat 全量微调.md | 0 ...wen-7B-Chat 接入langchain搭建知识库助手.md | 0 .../LLM.py | 0 .../creat_db.py | 0 .../readme.md | 0 .../run_gradio.py | 0 .../08-Qwen-7B-Chat Lora -4bit微调.ipynb | 0 .../08-Qwen-7B-Chat Lora -8bit微调.ipynb | 0 .../Qwen}/08-Qwen-7B-Chat Lora 低精度微调.md | 0 .../Qwen}/08-Qwen-7B-Chat Lora 低精度微调.py | 0 .../Qwen}/09-Qwen-1_8B-chat CPU 部署 .ipynb | 0 .../Qwen}/09-Qwen-1_8B-chat CPU 部署 .md | 0 {Qwen => models/Qwen}/environment.yml | 0 {Qwen => models/Qwen}/images/1.png | Bin {Qwen => models/Qwen}/images/2.png | Bin {Qwen => models/Qwen}/images/3.png | Bin {Qwen => models/Qwen}/images/4.png | Bin {Qwen => models/Qwen}/images/5.png | Bin {Qwen => models/Qwen}/images/6.png | Bin {Qwen => models/Qwen}/images/7.png | Bin {Qwen => models/Qwen}/images/8.png | Bin {Qwen => models/Qwen}/images/P-tuning.png | Bin .../01-Qwen1.5-7B-Chat FastApi 部署调用.md | 0 ...1.5-7B-Chat 接入langchain搭建知识库助手.md | 0 .../Qwen1.5}/03-Qwen1.5-7B-Chat WebDemo.md | 0 .../Qwen1.5}/04-Qwen1.5-7B-chat Lora 微调.md | 0 .../05-Qwen1.5-7B-Chat-GPTQ-Int4 WebDemo.md | 0 .../Qwen1.5}/06-Qwen1.5-MoE-A2.7B.md | 0 .../07-Qwen1.5-7B-Chat vLLM 推理部署调用.md | 0 ...Qwen1.5-7B-chat LoRA微调接入实验管理.ipynb | 0 ...08-Qwen1.5-7B-chat LoRA微调接入实验管理.md | 0 .../Qwen1.5}/Qwen1.5-7B-Chat Lora.ipynb | 0 .../Qwen1.5}/benchmark_throughput.py | 0 {Qwen1.5 => models/Qwen1.5}/images/2.png | Bin {Qwen1.5 => models/Qwen1.5}/images/6.png | Bin .../images/Qwen1.5-7b-gptq-int4-1.png | Bin .../images/Qwen1.5-7b-gptq-int4-2.png | Bin .../Qwen1.5}/images/Qwen1.5-vllm-api-stat.png | Bin .../images/Qwen1.5-vllm-gpu-select.png | Bin .../Qwen1.5}/images/Qwen1.5-vllm.png | Bin .../Qwen1.5}/images/Qwen2-Web1.png | Bin .../Qwen1.5}/images/Qwen2-Web2.png | Bin .../Qwen1.5}/images/image-2.png | Bin .../Qwen1.5}/images/question_to_the_Qwen2.png | Bin .../Qwen1.5}/images/swanlabcallbacks.png | Bin .../Qwen1.5}/images/swanlabchart.png | Bin .../Qwen1.5}/images/swanlabdisplay.png | Bin .../Qwen1.5}/images/swanlabsettings.png | Bin .../Qwen1.5}/images/swanlabweb.png | Bin .../01-Qwen2-7B-Instruct FastApi 部署调用.md | 0 .../02-Qwen2-7B-Instruct Langchain 接入.md | 0 .../03-Qwen2-7B-Instruct WebDemo部署.md | 342 ++++---- .../04-Qwen2-7B-Instruct vLLM 部署调用.md | 0 .../Qwen2}/05-Qwen2-7B-Instruct Lora 微调.md | 0 .../Qwen2}/05-Qwen2-7B-Instruct Lora.ipynb | 0 .../Qwen2}/benchmark_throughput.py | 0 {Qwen2 => models/Qwen2}/images/01-0.png | Bin {Qwen2 => models/Qwen2}/images/01-1.png | Bin {Qwen2 => models/Qwen2}/images/01-2.png | Bin {Qwen2 => models/Qwen2}/images/01-3.png | Bin {Qwen2 => models/Qwen2}/images/01-4.png | Bin {Qwen2 => models/Qwen2}/images/01-5.png | Bin {Qwen2 => models/Qwen2}/images/01-6.png | Bin {Qwen2 => models/Qwen2}/images/01-7.png | Bin {Qwen2 => models/Qwen2}/images/02-1.png | Bin {Qwen2 => models/Qwen2}/images/03-0.png | Bin {Qwen2 => models/Qwen2}/images/03-1.png | Bin {Qwen2 => models/Qwen2}/images/03-10.png | Bin {Qwen2 => models/Qwen2}/images/03-11.png | Bin {Qwen2 => models/Qwen2}/images/03-12.png | Bin {Qwen2 => models/Qwen2}/images/03-13.png | Bin {Qwen2 => models/Qwen2}/images/03-14.png | Bin {Qwen2 => models/Qwen2}/images/03-15.png | Bin {Qwen2 => models/Qwen2}/images/03-16.png | Bin {Qwen2 => models/Qwen2}/images/03-17.png | Bin {Qwen2 => models/Qwen2}/images/03-2.png | Bin {Qwen2 => models/Qwen2}/images/03-3.png | Bin {Qwen2 => models/Qwen2}/images/03-4.png | Bin {Qwen2 => models/Qwen2}/images/03-5.png | Bin {Qwen2 => models/Qwen2}/images/03-6.png | Bin {Qwen2 => models/Qwen2}/images/03-7.png | Bin {Qwen2 => models/Qwen2}/images/03-8.png | Bin {Qwen2 => models/Qwen2}/images/03-9.png | Bin {Qwen2 => models/Qwen2}/images/fig4-1.png | Bin {Qwen2 => models/Qwen2}/images/fig4-10.png | Bin {Qwen2 => models/Qwen2}/images/fig4-11.png | Bin {Qwen2 => models/Qwen2}/images/fig4-12.png | Bin {Qwen2 => models/Qwen2}/images/fig4-13.png | Bin {Qwen2 => models/Qwen2}/images/fig4-14.png | Bin {Qwen2 => models/Qwen2}/images/fig4-2.png | Bin {Qwen2 => models/Qwen2}/images/fig4-3.png | Bin {Qwen2 => models/Qwen2}/images/fig4-4.png | Bin {Qwen2 => models/Qwen2}/images/fig4-5.png | Bin {Qwen2 => models/Qwen2}/images/fig4-6.png | Bin {Qwen2 => models/Qwen2}/images/fig4-7.png | Bin {Qwen2 => models/Qwen2}/images/fig4-8.png | Bin {Qwen2 => models/Qwen2}/images/fig4-9.png | Bin .../01-TransNormerLLM-7B FastApi 部署调用.md | 0 ...ormerLLM-7B 接入langchain搭建知识库助手.md | 0 .../03-TransNormerLLM-7B WebDemo.md | 0 .../04-TransNormerLLM-7B-chat-Lora.ipynb | 0 .../04-TrasnNormerLLM-7B Lora 微调.md | 0 .../images/Jupyter-response.png | Bin .../TransNormerLLM}/images/Machine-Config.png | Bin .../images/TransNormer-structure.png | Bin .../images/python-terminal.png | Bin .../images/python-terminal2.png | Bin .../images/question_to_the_TransNormer.png | Bin .../TransNormerLLM}/images/response.png | Bin .../TransNormerLLM}/images/server-ok.png | Bin .../TransNormerLLM}/images/start-jupyter.png | Bin .../01-XVERSE-7B-chat Transformers推理.md | 0 .../XVERSE}/02-XVERSE-7B-chat FastAPI部署.md | 0 .../03-XVERSE-7B-chat langchain 接入.md | 0 .../XVERSE}/04-XVERSE-7B-chat WebDemo 部署.md | 0 .../XVERSE}/05-XVERSE-7B-Chat Lora 微调.ipynb | 0 .../XVERSE}/05-XVERSE-7B-Chat Lora 微调.md | 0 .../XVERSE}/06-XVERSE-MoE-A4.2B.md | 0 {XVERSE => models/XVERSE}/code/LLM.py | 0 {XVERSE => models/XVERSE}/code/api.py | 0 {XVERSE => models/XVERSE}/code/chatBot.py | 0 {XVERSE => models/XVERSE}/code/data_format.py | 0 .../XVERSE}/code/model_download.py | 0 .../XVERSE}/code/requirement.txt | 0 {XVERSE => models/XVERSE}/code/xverse.py | 0 {XVERSE => models/XVERSE}/images/1.png | Bin {XVERSE => models/XVERSE}/images/2.png | Bin {XVERSE => models/XVERSE}/images/3.png | Bin {XVERSE => models/XVERSE}/images/4.png | Bin {XVERSE => models/XVERSE}/images/5.png | Bin {XVERSE => models/XVERSE}/images/6.png | Bin .../Yi}/01-Yi-6B-Chat FastApi 部署调用.md | 324 ++++---- ...-Yi-6B-Chat 接入langchain搭建知识库助手.md | 754 +++++++++--------- {Yi => models/Yi}/03-Yi-6B-chat WebDemo.md | 0 {Yi => models/Yi}/04-Yi-6B-Chat Lora 微调.md | 0 {Yi => models/Yi}/04-Yi-6B-chat Lora微调.py | 0 {Yi => models/Yi}/images/1.png | Bin {Yi => models/Yi}/images/2.png | Bin {Yi => models/Yi}/images/3.png | Bin {Yi => models/Yi}/images/4.png | Bin {Yi => models/Yi}/images/5.png | Bin {Yi => models/Yi}/images/6.png | Bin {Yi => models/Yi}/images/Yi-Web1.png | Bin {Yi => models/Yi}/images/Yi-Web2.png | Bin .../Yi}/images/question_to_the_Yi.png | Bin .../Yi}/images/search_question_chain.png | Bin .../01-Yuan2.0-M32 FastApi 部署调用.md | 0 .../02-Yuan2.0-M32 Langchain 接入.md | 0 .../03-Yuan2.0-M32 WebDemo部署.md | 358 ++++----- {Yuan2.0-M32 => models/Yuan2.0-M32}/README.md | 0 .../Yuan2.0-M32}/images/01-1.png | Bin .../Yuan2.0-M32}/images/01-2.png | Bin .../Yuan2.0-M32}/images/01-3.png | Bin .../Yuan2.0-M32}/images/01-4-0.png | Bin .../Yuan2.0-M32}/images/01-4-1.png | Bin .../Yuan2.0-M32}/images/01-5.png | Bin .../Yuan2.0-M32}/images/01-6.png | Bin .../Yuan2.0-M32}/images/01-7.png | Bin .../Yuan2.0-M32}/images/02-0.png | Bin .../Yuan2.0-M32}/images/03-0.png | Bin .../Yuan2.0-M32}/images/03-1.png | Bin .../Yuan2.0-M32}/images/03-2.png | Bin .../Yuan2.0-M32}/images/03-3.png | Bin .../Yuan2.0-M32}/images/03-4.png | Bin .../Yuan2.0-M32}/images/autodl-fs.png | Bin .../Yuan2.0-M32}/images/gpu.png | Bin .../Yuan2.0-M32}/images/yuan2.0-m32-0.jpg | Bin .../Yuan2.0-M32}/images/yuan2.0-m32-1.jpg | Bin .../01-Yuan2.0-2B FastApi 部署调用.md | 0 .../Yuan2.0}/02-Yuan2.0-2B Langchain 接入.md | 0 .../Yuan2.0}/03-Yuan2.0-2B WebDemo部署.md | 314 ++++---- .../Yuan2.0}/04-Yuan2.0-2B vLLM部署调用.ipynb | 0 .../Yuan2.0}/04-Yuan2.0-2B vLLM部署调用.md | 0 .../Yuan2.0}/05-Yuan2.0-2B Lora-bf16.ipynb | 0 .../Yuan2.0}/05-Yuan2.0-2B Lora-fp16.ipynb | 0 .../Yuan2.0}/05-Yuan2.0-2B Lora微调.md | 0 {Yuan2.0 => models/Yuan2.0}/README.md | 0 {Yuan2.0 => models/Yuan2.0}/images/01-1.png | Bin {Yuan2.0 => models/Yuan2.0}/images/01-2.png | Bin {Yuan2.0 => models/Yuan2.0}/images/01-3.png | Bin {Yuan2.0 => models/Yuan2.0}/images/01-4-0.png | Bin {Yuan2.0 => models/Yuan2.0}/images/01-4-1.png | Bin {Yuan2.0 => models/Yuan2.0}/images/01-5.png | Bin {Yuan2.0 => models/Yuan2.0}/images/01-6.png | Bin {Yuan2.0 => models/Yuan2.0}/images/01-7.png | Bin {Yuan2.0 => models/Yuan2.0}/images/02-0.png | Bin {Yuan2.0 => models/Yuan2.0}/images/03-0.png | Bin {Yuan2.0 => models/Yuan2.0}/images/03-1.png | Bin {Yuan2.0 => models/Yuan2.0}/images/03-2.png | Bin {Yuan2.0 => models/Yuan2.0}/images/03-3.png | Bin {Yuan2.0 => models/Yuan2.0}/images/03-4.png | Bin {Yuan2.0 => models/Yuan2.0}/images/04-0.png | Bin {Yuan2.0 => models/Yuan2.0}/images/04-1.png | Bin .../Yuan2.0}/images/05-fp-0.png | Bin .../Yuan2.0}/images/05-fp-1.png | Bin .../Yuan2.0}/images/05-fp-2.png | Bin .../Yuan2.0}/images/05-fp-3.png | Bin .../Yuan2.0}/images/05-gpu-0.png | Bin .../Yuan2.0}/images/05-gpu-1.png | Bin .../Yuan2.0}/images/autodl-fs.png | Bin .../Yuan2.0}/images/yuan2.0-0.png | Bin .../Yuan2.0}/images/yuan2.0-1.jpg | Bin .../01-Index-1.9B-chat FastApi 部署调用.md | 0 .../02-Index-1.9B-Chat 接入 LangChain.md | 0 .../03-Index-1.9B-chat WebDemo部署.md | 290 +++---- .../04-Index-1.9B-Chat Lora 微调.md | 0 .../04-Index-1.9B-Chat Lora.ipynb | 0 .../bilibili_Index-1.9B}/images/01-1.png | Bin .../bilibili_Index-1.9B}/images/03-11.png | Bin .../bilibili_Index-1.9B}/images/03-12.png | Bin .../bilibili_Index-1.9B}/images/03-13.png | Bin .../bilibili_Index-1.9B}/images/03-14.png | Bin .../bilibili_Index-1.9B}/images/03-15.png | Bin .../bilibili_Index-1.9B}/images/03-16.png | Bin .../bilibili_Index-1.9B}/images/03-17.png | Bin .../bilibili_Index-1.9B}/images/fig4-1.png | Bin .../bilibili_Index-1.9B}/images/fig4-2.png | Bin .../bilibili_Index-1.9B}/images/fig4-3.png | Bin .../bilibili_Index-1.9B}/images/fig4-4.png | Bin .../bilibili_Index-1.9B}/images/image01-0.png | Bin .../bilibili_Index-1.9B}/images/image01-1.png | Bin .../bilibili_Index-1.9B}/images/image01-2.png | Bin .../bilibili_Index-1.9B}/images/image01-3.png | Bin .../bilibili_Index-1.9B}/images/image01-4.png | Bin .../bilibili_Index-1.9B}/images/image02-1.png | Bin .../bilibili_Index-1.9B}/images/image02-2.png | Bin ...Phi-3-mini-4k-instruct FastApi 部署调用.md | 338 ++++---- ...2-Phi-3-mini-4k-instruct langchain 接入.md | 280 +++---- .../03-Phi-3-mini-4k-instruct WebDemo部署.md | 0 .../04-Phi-3-mini-4k-Instruct Lora 微调.md | 0 .../phi-3}/Phi-3-mini-4k-Instruct-Lora.ipynb | 0 {phi-3 => models/phi-3}/assets/01-1.png | Bin {phi-3 => models/phi-3}/assets/02-1.png | Bin {phi-3 => models/phi-3}/assets/02-2.png | Bin {phi-3 => models/phi-3}/assets/03-1.png | Bin 518 files changed, 2632 insertions(+), 2619 deletions(-) create mode 100644 examples/readme.md rename {Atom => models/Atom}/01-Atom-7B-chat-WebDemo.md (100%) rename {Atom => models/Atom}/02-Atom-7B-Chat Lora 微调.md (100%) rename {Atom => models/Atom}/02-Atom-7B-Chat-Lora/train.py (100%) rename {Atom => models/Atom}/02-Atom-7B-Chat-Lora/train.sh (100%) rename {Atom => models/Atom}/03-Atom-7B-Chat 接入langchain搭建知识库助手.md (100%) rename {Atom => models/Atom}/03-Atom-7B-Chat 接入langchain搭建知识库助手/LLM.py (100%) rename {Atom => models/Atom}/03-Atom-7B-Chat 接入langchain搭建知识库助手/creat_db.py (100%) rename {Atom => models/Atom}/03-Atom-7B-Chat 接入langchain搭建知识库助手/readme.md (100%) rename {Atom => models/Atom}/03-Atom-7B-Chat 接入langchain搭建知识库助手/run_gradio.py (100%) rename {Atom => models/Atom}/04-Atom-7B-chat 全量微调.md (100%) rename {Atom => models/Atom}/images/image-1.png (100%) rename {Atom => models/Atom}/images/image-2.png (100%) rename {Atom => models/Atom}/images/image-3.png (100%) rename {Atom => models/Atom}/images/image-4.png (100%) rename {Atom => models/Atom}/images/image-5.png (100%) rename {Atom => models/Atom}/images/image-6.png (100%) rename {Atom => models/Atom}/images/image-7.png (100%) rename {Atom => models/Atom}/images/image-8.png (100%) rename {Atom => models/Atom}/images/image-9.png (100%) rename {BaiChuan => models/BaiChuan}/01-Baichuan2-7B-chat+FastApi+部署调用.md (100%) rename {BaiChuan => models/BaiChuan}/02-Baichuan-7B-chat+WebDemo.md (100%) rename {BaiChuan => models/BaiChuan}/03-Baichuan2-7B-chat接入LangChain框架.md (100%) rename {BaiChuan => models/BaiChuan}/04-Baichuan2-7B-chat Lora 微调.ipynb (100%) rename {BaiChuan => models/BaiChuan}/04-Baichuan2-7B-chat+lora+微调.md (100%) rename {BaiChuan => models/BaiChuan}/images/image1.png (100%) rename {BaiChuan => models/BaiChuan}/images/image10.png (100%) rename {BaiChuan => models/BaiChuan}/images/image11.png (100%) rename {BaiChuan => models/BaiChuan}/images/image12.png (100%) rename {BaiChuan => models/BaiChuan}/images/image13.png (100%) rename {BaiChuan => models/BaiChuan}/images/image14.png (100%) rename {BaiChuan => models/BaiChuan}/images/image15.png (100%) rename {BaiChuan => models/BaiChuan}/images/image16.png (100%) rename {BaiChuan => models/BaiChuan}/images/image17.png (100%) rename {BaiChuan => models/BaiChuan}/images/image18.png (100%) rename {BaiChuan => models/BaiChuan}/images/image2.png (100%) rename {BaiChuan => models/BaiChuan}/images/image20.png (100%) rename {BaiChuan => models/BaiChuan}/images/image23.png (100%) rename {BaiChuan => models/BaiChuan}/images/image25.png (100%) rename {BaiChuan => models/BaiChuan}/images/image26.png (100%) rename {BaiChuan => models/BaiChuan}/images/image27.png (100%) rename {BaiChuan => models/BaiChuan}/images/image3.png (100%) rename {BaiChuan => models/BaiChuan}/images/image4.png (100%) rename {BaiChuan => models/BaiChuan}/images/image6.png (100%) rename {BaiChuan => models/BaiChuan}/images/image7.png (100%) rename {BaiChuan => models/BaiChuan}/images/image8.png (100%) rename {BaiChuan => models/BaiChuan}/images/image9.png (100%) rename {BlueLM => models/BlueLM}/01-BlueLM-7B-Chat FastApi 部署.md (100%) rename {BlueLM => models/BlueLM}/02-BlueLM-7B-Chat langchain 接入.md (100%) rename {BlueLM => models/BlueLM}/03-BlueLM-7B-Chat WebDemo 部署.md (100%) rename {BlueLM => models/BlueLM}/04-BlueLM-7B-Chat Lora 微调.ipynb (100%) rename {BlueLM => models/BlueLM}/04-BlueLM-7B-Chat Lora 微调.md (100%) rename {BlueLM => models/BlueLM}/04-BlueLM-7B-Chat Lora 微调.py (100%) rename {BlueLM => models/BlueLM}/images/202403191628941.png (100%) rename {BlueLM => models/BlueLM}/images/202403191813385.png (100%) rename {BlueLM => models/BlueLM}/images/202403201210690.png (100%) rename {BlueLM => models/BlueLM}/images/202403201229542.png (100%) rename {BlueLM => models/BlueLM}/images/202403202153465.png (100%) rename {CharacterGLM => models/CharacterGLM}/01-CharacterGLM-6B Transformer部署调用.md (98%) rename {CharacterGLM => models/CharacterGLM}/02-CharacterGLM-6B FastApi部署调用.md (97%) rename {CharacterGLM => models/CharacterGLM}/03-CharacterGLM-6B-chat.md (97%) rename {CharacterGLM => models/CharacterGLM}/04-CharacterGLM-6B Lora微调.md (97%) rename {CharacterGLM => models/CharacterGLM}/04-CharacterGLM-6B-Lora微调.ipynb (100%) rename {CharacterGLM => models/CharacterGLM}/04-CharacterGLM-6B-Lora微调.py (100%) rename {CharacterGLM => models/CharacterGLM}/image/03-webdemo_show.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/03-修改路径.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/03-运行clidemo.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/03-运行webdemo.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/image-1.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/image-2.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/image-3.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/image-4.png (100%) rename {CharacterGLM => models/CharacterGLM}/image/readme.md (100%) rename {CharacterGLM => models/CharacterGLM}/readme.md (100%) rename {ChatGLM => models/ChatGLM}/01-ChatGLM3-6B Transformer部署调用.md (100%) rename {ChatGLM => models/ChatGLM}/02-ChatGLM3-6B FastApi部署调用.md (100%) rename {ChatGLM => models/ChatGLM}/03-ChatGLM3-6B-chat.md (100%) rename {ChatGLM => models/ChatGLM}/04-ChatGLM3-6B-Code-Interpreter.md (100%) rename {ChatGLM => models/ChatGLM}/05-ChatGLM3-6B接入LangChain搭建知识库助手.md (100%) rename {ChatGLM => models/ChatGLM}/05-ChatGLM3-6B接入LangChain搭建知识库助手/LLM.py (100%) rename {ChatGLM => models/ChatGLM}/05-ChatGLM3-6B接入LangChain搭建知识库助手/create_db.py (100%) rename {ChatGLM => models/ChatGLM}/05-ChatGLM3-6B接入LangChain搭建知识库助手/run_gradio.py (100%) rename {ChatGLM => models/ChatGLM}/06-ChatGLM3-6B-Lora微调.ipynb (100%) rename {ChatGLM => models/ChatGLM}/06-ChatGLM3-6B-Lora微调.md (100%) rename {ChatGLM => models/ChatGLM}/06-ChatGLM3-6B-Lora微调.py (100%) rename {ChatGLM => models/ChatGLM}/images/image-1.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-2.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-3.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-4.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-5.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-6.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-7.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-8.png (100%) rename {ChatGLM => models/ChatGLM}/images/image-9.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/01-DeepSeek-Coder-V2-Lite-Instruct FastApi 部署调用.md (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/02-DeepSeek-Coder-V2-Lite-Instruct 接入 LangChain.md (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/03-DeepSeek-Coder-V2-Lite-Instruct WebDemo 部署.md (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.ipynb (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.md (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-1.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-2.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-3.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-4.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-5.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-6.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-7.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-8.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig1-9.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig2-1.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig2-2.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/fig2-3.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/image03-1.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/image03-2.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/image03-3.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/image03-4.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/image03-5.png (100%) rename {DeepSeek-Coder-V2 => models/DeepSeek-Coder-V2}/images/image03-6.png (100%) rename {DeepSeek => models/DeepSeek}/01-DeepSeek-7B-chat FastApi.md (100%) rename {DeepSeek => models/DeepSeek}/02-DeepSeek-7B-chat langchain.md (100%) rename {DeepSeek => models/DeepSeek}/03-DeepSeek-7B-chat WebDemo.md (100%) rename {DeepSeek => models/DeepSeek}/04-DeepSeek-7B-chat Lora 微调.ipynb (100%) rename {DeepSeek => models/DeepSeek}/04-DeepSeek-7B-chat Lora 微调.md (100%) rename {DeepSeek => models/DeepSeek}/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.ipynb (100%) rename {DeepSeek => models/DeepSeek}/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.md (100%) rename {DeepSeek => models/DeepSeek}/06-DeepSeek-MoE-16b-chat FastApi部署调用.md (100%) rename {DeepSeek => models/DeepSeek}/06-DeepSeek-MoE-16b-chat Transformer部署调用.md (100%) rename {DeepSeek => models/DeepSeek}/07-deepseek_fine_tune.ipynb (100%) rename {DeepSeek => models/DeepSeek}/08-deepseek_web_demo.ipynb (100%) rename {DeepSeek => models/DeepSeek}/images/image-1.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-2.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-3.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-4.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-5.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-6.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-7.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-8.png (100%) rename {DeepSeek => models/DeepSeek}/images/image-9.png (100%) rename {GLM-4 => models/GLM-4}/01-GLM-4-9B-chat FastApi 部署调用.md (100%) rename {GLM-4 => models/GLM-4}/02-GLM-4-9B-chat langchain 接入.md (100%) rename {GLM-4 => models/GLM-4}/03-GLM-4-9B-Chat WebDemo.md (100%) rename {GLM-4 => models/GLM-4}/04-GLM-4-9B-Chat vLLM 部署调用.md (100%) rename {GLM-4 => models/GLM-4}/05-GLM-4-9B-chat Lora 微调.ipynb (100%) rename {GLM-4 => models/GLM-4}/05-GLM-4-9B-chat Lora 微调.md (100%) rename {GLM-4 => models/GLM-4}/benchmark_throughput.py (100%) rename {GLM-4 => models/GLM-4}/images/image-1.png (100%) rename {GLM-4 => models/GLM-4}/images/image01-1.png (100%) rename {GLM-4 => models/GLM-4}/images/image01-2.png (100%) rename {GLM-4 => models/GLM-4}/images/image01-3.png (100%) rename {GLM-4 => models/GLM-4}/images/image01-4.png (100%) rename {GLM-4 => models/GLM-4}/images/image01-5.png (100%) rename {GLM-4 => models/GLM-4}/images/image02-1.png (100%) rename {GLM-4 => models/GLM-4}/images/image03-1.png (100%) rename {GLM-4 => models/GLM-4}/images/image03-2.png (100%) rename {GLM-4 => models/GLM-4}/images/image04-1.png (100%) rename {Gemma => models/Gemma}/01-Gemma-2B-Instruct FastApi 部署调用.md (100%) rename {Gemma => models/Gemma}/02-Gemma-2B-Instruct langchain 接入.md (100%) rename {Gemma => models/Gemma}/03-Gemma-2B-Instruct WebDemo 部署.md (100%) rename {Gemma => models/Gemma}/04-Gemma-2B-Instruct Lora微调.md (100%) rename {Gemma => models/Gemma}/04-Gemma-2B-Lora微调.ipynb (100%) rename {Gemma => models/Gemma}/images/image-1.png (100%) rename {Gemma => models/Gemma}/images/image-2.png (100%) rename {Gemma => models/Gemma}/images/image-3.png (100%) rename {Gemma => models/Gemma}/images/image-4.png (100%) rename {Gemma => models/Gemma}/images/image-5.png (100%) rename {Gemma2 => models/Gemma2}/01-Gemma-2-9b-it FastApi 部署调用.md (100%) rename {Gemma2 => models/Gemma2}/02-Gemma-2-9b-it langchain 接入.md (100%) rename {Gemma2 => models/Gemma2}/03-Gemma-2-9b-it WebDemo 部署.md (100%) rename {Gemma2 => models/Gemma2}/04-Gemma-2-9b-it peft lora微调.ipynb (100%) rename {Gemma2 => models/Gemma2}/04-Gemma-2-9b-it peft lora微调.md (100%) rename {Gemma2 => models/Gemma2}/images/01-1.png (100%) rename {Gemma2 => models/Gemma2}/images/01-4-0.png (100%) rename {Gemma2 => models/Gemma2}/images/01-4-1.png (100%) rename {Gemma2 => models/Gemma2}/images/01-5.png (100%) rename {Gemma2 => models/Gemma2}/images/01-6.png (100%) rename {Gemma2 => models/Gemma2}/images/01-7.png (100%) rename {Gemma2 => models/Gemma2}/images/02-1.png (100%) rename {Gemma2 => models/Gemma2}/images/03-0.png (100%) rename {Gemma2 => models/Gemma2}/images/03-1.png (100%) rename {Gemma2 => models/Gemma2}/images/03-2.png (100%) rename {Gemma2 => models/Gemma2}/images/03-3.png (100%) rename {Gemma2 => models/Gemma2}/images/04-1.png (100%) rename {Gemma2 => models/Gemma2}/images/04-2.png (100%) rename {General-Setting => models/General-Setting}/01-pip、conda换源.md (100%) rename {General-Setting => models/General-Setting}/02-AutoDL开放端口.md (100%) rename {General-Setting => models/General-Setting}/03-模型下载.md (100%) rename {General-Setting => models/General-Setting}/04-Issue&PR&update.md (100%) rename {General-Setting => models/General-Setting}/pic/Issue1.png (100%) rename {General-Setting => models/General-Setting}/pic/Issue2.png (100%) rename {General-Setting => models/General-Setting}/pic/PR.png (100%) rename {General-Setting => models/General-Setting}/pic/PR1.png (100%) rename {General-Setting => models/General-Setting}/pic/PR3.png (100%) rename {General-Setting => models/General-Setting}/pic/PR4.png (100%) rename {General-Setting => models/General-Setting}/pic/PR5.png (100%) rename {General-Setting => models/General-Setting}/pic/PR6.png (100%) rename {General-Setting => models/General-Setting}/pic/端口映射.png (100%) rename {InternLM => models/InternLM}/01-InternLM-Chat-7B Transformers 部署调用.md (100%) rename {InternLM => models/InternLM}/02-internLM-Chat-7B FastApi.md (100%) rename {InternLM => models/InternLM}/03-InternLM-Chat-7B.md (100%) rename {InternLM => models/InternLM}/04-Lagent+InternLM-Chat-7B-V1.1.md (100%) rename {InternLM => models/InternLM}/05-浦语灵笔图文理解&创作.md (100%) rename {InternLM => models/InternLM}/06-InternLM接入LangChain搭建知识库助手.md (100%) rename {InternLM => models/InternLM}/06-InternLM接入LangChain搭建知识库助手/LLM.py (100%) rename {InternLM => models/InternLM}/06-InternLM接入LangChain搭建知识库助手/creat_db.py (100%) rename {InternLM => models/InternLM}/06-InternLM接入LangChain搭建知识库助手/readme.md (100%) rename {InternLM => models/InternLM}/06-InternLM接入LangChain搭建知识库助手/run_gradio.py (100%) rename {InternLM => models/InternLM}/images/image-1.png (100%) rename {InternLM => models/InternLM}/images/image-10.png (100%) rename {InternLM => models/InternLM}/images/image-11.png (100%) rename {InternLM => models/InternLM}/images/image-12.png (100%) rename {InternLM => models/InternLM}/images/image-13.png (100%) rename {InternLM => models/InternLM}/images/image-14.png (100%) rename {InternLM => models/InternLM}/images/image-2.png (100%) rename {InternLM => models/InternLM}/images/image-3.png (100%) rename {InternLM => models/InternLM}/images/image-4.png (100%) rename {InternLM => models/InternLM}/images/image-5.png (100%) rename {InternLM => models/InternLM}/images/image-6.png (100%) rename {InternLM => models/InternLM}/images/image-7.png (100%) rename {InternLM => models/InternLM}/images/image-8.png (100%) rename {InternLM => models/InternLM}/images/image-9.png (100%) rename {InternLM => models/InternLM}/images/image.png (100%) rename {InternLM2 => models/InternLM2}/01-InternLM2-7B-chat FastAPI部署.md (100%) rename {InternLM2 => models/InternLM2}/02-InternLM2-7B-chat langchain 接入.md (100%) rename {InternLM2 => models/InternLM2}/03-InternLM2-7B-chat WebDemo 部署.md (97%) rename {InternLM2 => models/InternLM2}/04-InternLM2-7B-chat Xtuner Qlora 微调.md (97%) rename {InternLM2 => models/InternLM2}/dataset/心理大模型-职场焦虑语料.xlsx (100%) rename {InternLM2 => models/InternLM2}/images/1.png (100%) rename {InternLM2 => models/InternLM2}/images/2.png (100%) rename {InternLM2 => models/InternLM2}/images/3-1.png (100%) rename {InternLM2 => models/InternLM2}/images/3-2.png (100%) rename {InternLM2 => models/InternLM2}/images/3-3.png (100%) rename {InternLM2 => models/InternLM2}/images/3-4.png (100%) rename {InternLM2 => models/InternLM2}/images/3-5.png (100%) rename {InternLM2 => models/InternLM2}/images/3-6.png (100%) rename {InternLM2 => models/InternLM2}/images/3-7.png (100%) rename {InternLM2 => models/InternLM2}/images/3-8.png (100%) rename {InternLM2 => models/InternLM2}/images/3.png (100%) rename {InternLM2 => models/InternLM2}/images/4-1.png (100%) rename {InternLM2 => models/InternLM2}/images/4-2.png (100%) rename {InternLM2 => models/InternLM2}/images/4-3.png (100%) rename {InternLM2 => models/InternLM2}/images/4-4.png (100%) rename {LLaMA3 => models/LLaMA3}/01-LLaMA3-8B-Instruct FastApi 部署调用.md (100%) rename {LLaMA3 => models/LLaMA3}/02-LLaMA3-8B-Instruct langchain 接入.md (100%) rename {LLaMA3 => models/LLaMA3}/03-LLaMA3-8B-Instruct WebDemo 部署.md (100%) rename {LLaMA3 => models/LLaMA3}/04-LLaMA3-8B-Instruct Lora 微调.md (100%) rename {LLaMA3 => models/LLaMA3}/LLaMA3-8B-Instruct Lora.ipynb (100%) rename {LLaMA3 => models/LLaMA3}/images/api_resp.png (100%) rename {LLaMA3 => models/LLaMA3}/images/api_start.png (100%) rename {LLaMA3 => models/LLaMA3}/images/image-1.png (100%) rename {LLaMA3 => models/LLaMA3}/images/image-2.png (100%) rename {LLaMA3 => models/LLaMA3}/images/image-3.png (100%) rename {MiniCPM => models/MiniCPM}/MiniCPM-2B-chat FastApi 部署调用.md (100%) rename {MiniCPM => models/MiniCPM}/MiniCPM-2B-chat Lora && Full 微调.md (100%) rename {MiniCPM => models/MiniCPM}/MiniCPM-2B-chat WebDemo部署.md (100%) rename {MiniCPM => models/MiniCPM}/MiniCPM-2B-chat langchain接入.md (100%) rename {MiniCPM => models/MiniCPM}/MiniCPM-2B-chat transformers 部署调用.md (100%) rename {MiniCPM => models/MiniCPM}/ds_config.json (100%) rename {MiniCPM => models/MiniCPM}/images/image-1.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-10.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-2.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-3.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-4.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-5.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-6.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-7.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-8.png (100%) rename {MiniCPM => models/MiniCPM}/images/image-9.png (100%) rename {MiniCPM => models/MiniCPM}/train.py (100%) rename {MiniCPM => models/MiniCPM}/train.sh (100%) rename {Qwen-Audio => models/Qwen-Audio}/01-Qwen-Audio-chat FastApi.md (100%) rename {Qwen-Audio => models/Qwen-Audio}/02-Qwen-Audio-chat WebDemo.md (100%) rename {Qwen-Audio => models/Qwen-Audio}/images/image-1.png (100%) rename {Qwen-Audio => models/Qwen-Audio}/images/image-2.png (100%) rename {Qwen-Audio => models/Qwen-Audio}/images/image-3.png (100%) rename {Qwen-Audio => models/Qwen-Audio}/images/image-4.png (100%) rename {Qwen => models/Qwen}/01-Qwen-7B-Chat Transformers部署调用.md (100%) rename {Qwen => models/Qwen}/02-Qwen-7B-Chat FastApi 部署调用.md (100%) rename {Qwen => models/Qwen}/03-Qwen-7B-Chat WebDemo.md (100%) rename {Qwen => models/Qwen}/04-Qwen-7B-Chat Lora 微调.ipynb (100%) rename {Qwen => models/Qwen}/04-Qwen-7B-Chat Lora 微调.md (100%) rename {Qwen => models/Qwen}/04-Qwen-7B-Chat Lora 微调.py (100%) rename {Qwen => models/Qwen}/05-Qwen-7B-Chat Ptuning 微调.md (100%) rename {Qwen => models/Qwen}/05-Qwen-7B-Chat Ptuning 微调.py (100%) rename {Qwen => models/Qwen}/06-Qwen-7B-chat 全量微调.md (100%) rename {Qwen => models/Qwen}/07-Qwen-7B-Chat 接入langchain搭建知识库助手.md (100%) rename {Qwen => models/Qwen}/07-Qwen-7B-Chat 接入langchain搭建知识库助手/LLM.py (100%) rename {Qwen => models/Qwen}/07-Qwen-7B-Chat 接入langchain搭建知识库助手/creat_db.py (100%) rename {Qwen => models/Qwen}/07-Qwen-7B-Chat 接入langchain搭建知识库助手/readme.md (100%) rename {Qwen => models/Qwen}/07-Qwen-7B-Chat 接入langchain搭建知识库助手/run_gradio.py (100%) rename {Qwen => models/Qwen}/08-Qwen-7B-Chat Lora -4bit微调.ipynb (100%) rename {Qwen => models/Qwen}/08-Qwen-7B-Chat Lora -8bit微调.ipynb (100%) rename {Qwen => models/Qwen}/08-Qwen-7B-Chat Lora 低精度微调.md (100%) rename {Qwen => models/Qwen}/08-Qwen-7B-Chat Lora 低精度微调.py (100%) rename {Qwen => models/Qwen}/09-Qwen-1_8B-chat CPU 部署 .ipynb (100%) rename {Qwen => models/Qwen}/09-Qwen-1_8B-chat CPU 部署 .md (100%) rename {Qwen => models/Qwen}/environment.yml (100%) rename {Qwen => models/Qwen}/images/1.png (100%) rename {Qwen => models/Qwen}/images/2.png (100%) rename {Qwen => models/Qwen}/images/3.png (100%) rename {Qwen => models/Qwen}/images/4.png (100%) rename {Qwen => models/Qwen}/images/5.png (100%) rename {Qwen => models/Qwen}/images/6.png (100%) rename {Qwen => models/Qwen}/images/7.png (100%) rename {Qwen => models/Qwen}/images/8.png (100%) rename {Qwen => models/Qwen}/images/P-tuning.png (100%) rename {Qwen1.5 => models/Qwen1.5}/01-Qwen1.5-7B-Chat FastApi 部署调用.md (100%) rename {Qwen1.5 => models/Qwen1.5}/02-Qwen1.5-7B-Chat 接入langchain搭建知识库助手.md (100%) rename {Qwen1.5 => models/Qwen1.5}/03-Qwen1.5-7B-Chat WebDemo.md (100%) rename {Qwen1.5 => models/Qwen1.5}/04-Qwen1.5-7B-chat Lora 微调.md (100%) rename {Qwen1.5 => models/Qwen1.5}/05-Qwen1.5-7B-Chat-GPTQ-Int4 WebDemo.md (100%) rename {Qwen1.5 => models/Qwen1.5}/06-Qwen1.5-MoE-A2.7B.md (100%) rename {Qwen1.5 => models/Qwen1.5}/07-Qwen1.5-7B-Chat vLLM 推理部署调用.md (100%) rename {Qwen1.5 => models/Qwen1.5}/08-Qwen1.5-7B-chat LoRA微调接入实验管理.ipynb (100%) rename {Qwen1.5 => models/Qwen1.5}/08-Qwen1.5-7B-chat LoRA微调接入实验管理.md (100%) rename {Qwen1.5 => models/Qwen1.5}/Qwen1.5-7B-Chat Lora.ipynb (100%) rename {Qwen1.5 => models/Qwen1.5}/benchmark_throughput.py (100%) rename {Qwen1.5 => models/Qwen1.5}/images/2.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/6.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/Qwen1.5-7b-gptq-int4-1.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/Qwen1.5-7b-gptq-int4-2.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/Qwen1.5-vllm-api-stat.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/Qwen1.5-vllm-gpu-select.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/Qwen1.5-vllm.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/Qwen2-Web1.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/Qwen2-Web2.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/image-2.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/question_to_the_Qwen2.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/swanlabcallbacks.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/swanlabchart.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/swanlabdisplay.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/swanlabsettings.png (100%) rename {Qwen1.5 => models/Qwen1.5}/images/swanlabweb.png (100%) rename {Qwen2 => models/Qwen2}/01-Qwen2-7B-Instruct FastApi 部署调用.md (100%) rename {Qwen2 => models/Qwen2}/02-Qwen2-7B-Instruct Langchain 接入.md (100%) rename {Qwen2 => models/Qwen2}/03-Qwen2-7B-Instruct WebDemo部署.md (97%) rename {Qwen2 => models/Qwen2}/04-Qwen2-7B-Instruct vLLM 部署调用.md (100%) rename {Qwen2 => models/Qwen2}/05-Qwen2-7B-Instruct Lora 微调.md (100%) rename {Qwen2 => models/Qwen2}/05-Qwen2-7B-Instruct Lora.ipynb (100%) rename {Qwen2 => models/Qwen2}/benchmark_throughput.py (100%) rename {Qwen2 => models/Qwen2}/images/01-0.png (100%) rename {Qwen2 => models/Qwen2}/images/01-1.png (100%) rename {Qwen2 => models/Qwen2}/images/01-2.png (100%) rename {Qwen2 => models/Qwen2}/images/01-3.png (100%) rename {Qwen2 => models/Qwen2}/images/01-4.png (100%) rename {Qwen2 => models/Qwen2}/images/01-5.png (100%) rename {Qwen2 => models/Qwen2}/images/01-6.png (100%) rename {Qwen2 => models/Qwen2}/images/01-7.png (100%) rename {Qwen2 => models/Qwen2}/images/02-1.png (100%) rename {Qwen2 => models/Qwen2}/images/03-0.png (100%) rename {Qwen2 => models/Qwen2}/images/03-1.png (100%) rename {Qwen2 => models/Qwen2}/images/03-10.png (100%) rename {Qwen2 => models/Qwen2}/images/03-11.png (100%) rename {Qwen2 => models/Qwen2}/images/03-12.png (100%) rename {Qwen2 => models/Qwen2}/images/03-13.png (100%) rename {Qwen2 => models/Qwen2}/images/03-14.png (100%) rename {Qwen2 => models/Qwen2}/images/03-15.png (100%) rename {Qwen2 => models/Qwen2}/images/03-16.png (100%) rename {Qwen2 => models/Qwen2}/images/03-17.png (100%) rename {Qwen2 => models/Qwen2}/images/03-2.png (100%) rename {Qwen2 => models/Qwen2}/images/03-3.png (100%) rename {Qwen2 => models/Qwen2}/images/03-4.png (100%) rename {Qwen2 => models/Qwen2}/images/03-5.png (100%) rename {Qwen2 => models/Qwen2}/images/03-6.png (100%) rename {Qwen2 => models/Qwen2}/images/03-7.png (100%) rename {Qwen2 => models/Qwen2}/images/03-8.png (100%) rename {Qwen2 => models/Qwen2}/images/03-9.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-1.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-10.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-11.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-12.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-13.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-14.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-2.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-3.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-4.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-5.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-6.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-7.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-8.png (100%) rename {Qwen2 => models/Qwen2}/images/fig4-9.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/01-TransNormerLLM-7B FastApi 部署调用.md (100%) rename {TransNormerLLM => models/TransNormerLLM}/02-TransNormerLLM-7B 接入langchain搭建知识库助手.md (100%) rename {TransNormerLLM => models/TransNormerLLM}/03-TransNormerLLM-7B WebDemo.md (100%) rename {TransNormerLLM => models/TransNormerLLM}/04-TransNormerLLM-7B-chat-Lora.ipynb (100%) rename {TransNormerLLM => models/TransNormerLLM}/04-TrasnNormerLLM-7B Lora 微调.md (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/Jupyter-response.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/Machine-Config.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/TransNormer-structure.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/python-terminal.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/python-terminal2.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/question_to_the_TransNormer.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/response.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/server-ok.png (100%) rename {TransNormerLLM => models/TransNormerLLM}/images/start-jupyter.png (100%) rename {XVERSE => models/XVERSE}/01-XVERSE-7B-chat Transformers推理.md (100%) rename {XVERSE => models/XVERSE}/02-XVERSE-7B-chat FastAPI部署.md (100%) rename {XVERSE => models/XVERSE}/03-XVERSE-7B-chat langchain 接入.md (100%) rename {XVERSE => models/XVERSE}/04-XVERSE-7B-chat WebDemo 部署.md (100%) rename {XVERSE => models/XVERSE}/05-XVERSE-7B-Chat Lora 微调.ipynb (100%) rename {XVERSE => models/XVERSE}/05-XVERSE-7B-Chat Lora 微调.md (100%) rename {XVERSE => models/XVERSE}/06-XVERSE-MoE-A4.2B.md (100%) rename {XVERSE => models/XVERSE}/code/LLM.py (100%) rename {XVERSE => models/XVERSE}/code/api.py (100%) rename {XVERSE => models/XVERSE}/code/chatBot.py (100%) rename {XVERSE => models/XVERSE}/code/data_format.py (100%) rename {XVERSE => models/XVERSE}/code/model_download.py (100%) rename {XVERSE => models/XVERSE}/code/requirement.txt (100%) rename {XVERSE => models/XVERSE}/code/xverse.py (100%) rename {XVERSE => models/XVERSE}/images/1.png (100%) rename {XVERSE => models/XVERSE}/images/2.png (100%) rename {XVERSE => models/XVERSE}/images/3.png (100%) rename {XVERSE => models/XVERSE}/images/4.png (100%) rename {XVERSE => models/XVERSE}/images/5.png (100%) rename {XVERSE => models/XVERSE}/images/6.png (100%) rename {Yi => models/Yi}/01-Yi-6B-Chat FastApi 部署调用.md (97%) rename {Yi => models/Yi}/02-Yi-6B-Chat 接入langchain搭建知识库助手.md (97%) rename {Yi => models/Yi}/03-Yi-6B-chat WebDemo.md (100%) rename {Yi => models/Yi}/04-Yi-6B-Chat Lora 微调.md (100%) rename {Yi => models/Yi}/04-Yi-6B-chat Lora微调.py (100%) rename {Yi => models/Yi}/images/1.png (100%) rename {Yi => models/Yi}/images/2.png (100%) rename {Yi => models/Yi}/images/3.png (100%) rename {Yi => models/Yi}/images/4.png (100%) rename {Yi => models/Yi}/images/5.png (100%) rename {Yi => models/Yi}/images/6.png (100%) rename {Yi => models/Yi}/images/Yi-Web1.png (100%) rename {Yi => models/Yi}/images/Yi-Web2.png (100%) rename {Yi => models/Yi}/images/question_to_the_Yi.png (100%) rename {Yi => models/Yi}/images/search_question_chain.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/01-Yuan2.0-M32 FastApi 部署调用.md (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/02-Yuan2.0-M32 Langchain 接入.md (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/03-Yuan2.0-M32 WebDemo部署.md (97%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/README.md (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-1.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-2.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-3.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-4-0.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-4-1.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-5.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-6.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/01-7.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/02-0.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/03-0.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/03-1.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/03-2.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/03-3.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/03-4.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/autodl-fs.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/gpu.png (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/yuan2.0-m32-0.jpg (100%) rename {Yuan2.0-M32 => models/Yuan2.0-M32}/images/yuan2.0-m32-1.jpg (100%) rename {Yuan2.0 => models/Yuan2.0}/01-Yuan2.0-2B FastApi 部署调用.md (100%) rename {Yuan2.0 => models/Yuan2.0}/02-Yuan2.0-2B Langchain 接入.md (100%) rename {Yuan2.0 => models/Yuan2.0}/03-Yuan2.0-2B WebDemo部署.md (97%) rename {Yuan2.0 => models/Yuan2.0}/04-Yuan2.0-2B vLLM部署调用.ipynb (100%) rename {Yuan2.0 => models/Yuan2.0}/04-Yuan2.0-2B vLLM部署调用.md (100%) rename {Yuan2.0 => models/Yuan2.0}/05-Yuan2.0-2B Lora-bf16.ipynb (100%) rename {Yuan2.0 => models/Yuan2.0}/05-Yuan2.0-2B Lora-fp16.ipynb (100%) rename {Yuan2.0 => models/Yuan2.0}/05-Yuan2.0-2B Lora微调.md (100%) rename {Yuan2.0 => models/Yuan2.0}/README.md (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-1.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-2.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-3.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-4-0.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-4-1.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-5.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-6.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/01-7.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/02-0.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/03-0.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/03-1.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/03-2.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/03-3.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/03-4.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/04-0.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/04-1.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/05-fp-0.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/05-fp-1.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/05-fp-2.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/05-fp-3.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/05-gpu-0.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/05-gpu-1.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/autodl-fs.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/yuan2.0-0.png (100%) rename {Yuan2.0 => models/Yuan2.0}/images/yuan2.0-1.jpg (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/01-Index-1.9B-chat FastApi 部署调用.md (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/02-Index-1.9B-Chat 接入 LangChain.md (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/03-Index-1.9B-chat WebDemo部署.md (97%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/04-Index-1.9B-Chat Lora 微调.md (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/04-Index-1.9B-Chat Lora.ipynb (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/01-1.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/03-11.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/03-12.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/03-13.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/03-14.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/03-15.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/03-16.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/03-17.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/fig4-1.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/fig4-2.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/fig4-3.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/fig4-4.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/image01-0.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/image01-1.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/image01-2.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/image01-3.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/image01-4.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/image02-1.png (100%) rename {bilibili_Index-1.9B => models/bilibili_Index-1.9B}/images/image02-2.png (100%) rename {phi-3 => models/phi-3}/01-Phi-3-mini-4k-instruct FastApi 部署调用.md (97%) rename {phi-3 => models/phi-3}/02-Phi-3-mini-4k-instruct langchain 接入.md (97%) rename {phi-3 => models/phi-3}/03-Phi-3-mini-4k-instruct WebDemo部署.md (100%) rename {phi-3 => models/phi-3}/04-Phi-3-mini-4k-Instruct Lora 微调.md (100%) rename {phi-3 => models/phi-3}/Phi-3-mini-4k-Instruct-Lora.ipynb (100%) rename {phi-3 => models/phi-3}/assets/01-1.png (100%) rename {phi-3 => models/phi-3}/assets/02-1.png (100%) rename {phi-3 => models/phi-3}/assets/02-2.png (100%) rename {phi-3 => models/phi-3}/assets/03-1.png (100%) diff --git a/README.md b/README.md index 34a263a..cc1c296 100644 --- a/README.md +++ b/README.md @@ -56,190 +56,190 @@ ### 已支持模型 - [Gemma-2-9b-it](https://huggingface.co/google/gemma-2-9b-it) - - [x] [Gemma-2-9b-it FastApi 部署调用](./Gemma2/01-Gemma-2-9b-it%20FastApi%20部署调用.md) @不要葱姜蒜 - - [x] [Gemma-2-9b-it langchain 接入](./Gemma2/02-Gemma-2-9b-it%20langchain%20接入.md) @不要葱姜蒜 - - [x] [Gemma-2-9b-it WebDemo 部署](./Gemma2/03-Gemma-2-9b-it%20WebDemo%20部署.md) @不要葱姜蒜 - - [x] [Gemma-2-9b-it Peft Lora 微调](./Gemma2/04-Gemma-2-9b-it%20peft%20lora微调.md) @不要葱姜蒜 + - [x] [Gemma-2-9b-it FastApi 部署调用](./models/Gemma2/01-Gemma-2-9b-it%20FastApi%20部署调用.md) @不要葱姜蒜 + - [x] [Gemma-2-9b-it langchain 接入](./models/Gemma2/02-Gemma-2-9b-it%20langchain%20接入.md) @不要葱姜蒜 + - [x] [Gemma-2-9b-it WebDemo 部署](./models/Gemma2/03-Gemma-2-9b-it%20WebDemo%20部署.md) @不要葱姜蒜 + - [x] [Gemma-2-9b-it Peft Lora 微调](./models/Gemma2/04-Gemma-2-9b-it%20peft%20lora微调.md) @不要葱姜蒜 - [Yuan2.0](https://github.com/IEIT-Yuan/Yuan-2.0) - - [x] [Yuan2.0-2B FastApi 部署调用](./Yuan2.0/01-Yuan2.0-2B%20FastApi%20部署调用.md) @张帆 - - [x] [Yuan2.0-2B Langchain 接入](./Yuan2.0/02-Yuan2.0-2B%20Langchain%20接入.md) @张帆 - - [x] [Yuan2.0-2B WebDemo部署](./Yuan2.0/03-Yuan2.0-2B%20WebDemo部署.md) @张帆 - - [x] [Yuan2.0-2B vLLM部署调用](./Yuan2.0/04-Yuan2.0-2B%20vLLM部署调用.md) @张帆 - - [x] [Yuan2.0-2B Lora微调](./Yuan2.0/05-Yuan2.0-2B%20Lora微调.md) @张帆 + - [x] [Yuan2.0-2B FastApi 部署调用](./models/Yuan2.0/01-Yuan2.0-2B%20FastApi%20部署调用.md) @张帆 + - [x] [Yuan2.0-2B Langchain 接入](./models/Yuan2.0/02-Yuan2.0-2B%20Langchain%20接入.md) @张帆 + - [x] [Yuan2.0-2B WebDemo部署](./models/Yuan2.0/03-Yuan2.0-2B%20WebDemo部署.md) @张帆 + - [x] [Yuan2.0-2B vLLM部署调用](./models/Yuan2.0/04-Yuan2.0-2B%20vLLM部署调用.md) @张帆 + - [x] [Yuan2.0-2B Lora微调](./models/Yuan2.0/05-Yuan2.0-2B%20Lora微调.md) @张帆 - [Yuan2.0-M32](https://github.com/IEIT-Yuan/Yuan2.0-M32) - - [x] [Yuan2.0-M32 FastApi 部署调用](./Yuan2.0-M32/01-Yuan2.0-M32%20FastApi%20部署调用.md) @张帆 - - [x] [Yuan2.0-M32 Langchain 接入](./Yuan2.0-M32/02-Yuan2.0-M32%20Langchain%20接入.md) @张帆 - - [x] [Yuan2.0-M32 WebDemo部署](./Yuan2.0-M32/03-Yuan2.0-M32%20WebDemo部署.md) @张帆 + - [x] [Yuan2.0-M32 FastApi 部署调用](./models/Yuan2.0-M32/01-Yuan2.0-M32%20FastApi%20部署调用.md) @张帆 + - [x] [Yuan2.0-M32 Langchain 接入](./models/Yuan2.0-M32/02-Yuan2.0-M32%20Langchain%20接入.md) @张帆 + - [x] [Yuan2.0-M32 WebDemo部署](./models/Yuan2.0-M32/03-Yuan2.0-M32%20WebDemo部署.md) @张帆 - [DeepSeek-Coder-V2](https://github.com/deepseek-ai/DeepSeek-Coder-V2) - - [x] [DeepSeek-Coder-V2-Lite-Instruct FastApi 部署调用](./DeepSeek-Coder-V2/01-DeepSeek-Coder-V2-Lite-Instruct%20FastApi%20部署调用.md) @姜舒凡 - - [x] [DeepSeek-Coder-V2-Lite-Instruct langchain 接入](./DeepSeek-Coder-V2/02-DeepSeek-Coder-V2-Lite-Instruct%20接入%20LangChain.md) @姜舒凡 - - [x] [DeepSeek-Coder-V2-Lite-Instruct WebDemo 部署](./DeepSeek-Coder-V2/03-DeepSeek-Coder-V2-Lite-Instruct%20WebDemo%20部署.md) @Kailigithub - - [x] [DeepSeek-Coder-V2-Lite-Instruct Lora 微调](./DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct%20Lora%20微调.md) @余洋 + - [x] [DeepSeek-Coder-V2-Lite-Instruct FastApi 部署调用](./models/DeepSeek-Coder-V2/01-DeepSeek-Coder-V2-Lite-Instruct%20FastApi%20部署调用.md) @姜舒凡 + - [x] [DeepSeek-Coder-V2-Lite-Instruct langchain 接入](./models/DeepSeek-Coder-V2/02-DeepSeek-Coder-V2-Lite-Instruct%20接入%20LangChain.md) @姜舒凡 + - [x] [DeepSeek-Coder-V2-Lite-Instruct WebDemo 部署](./models/DeepSeek-Coder-V2/03-DeepSeek-Coder-V2-Lite-Instruct%20WebDemo%20部署.md) @Kailigithub + - [x] [DeepSeek-Coder-V2-Lite-Instruct Lora 微调](./models/DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct%20Lora%20微调.md) @余洋 - [哔哩哔哩 Index-1.9B](https://github.com/bilibili/Index-1.9B) - - [x] [Index-1.9B-Chat FastApi 部署调用](./bilibili_Index-1.9B/01-Index-1.9B-chat%20FastApi%20部署调用.md) @邓恺俊 - - [x] [Index-1.9B-Chat langchain 接入](./bilibili_Index-1.9B/02-Index-1.9B-Chat%20接入%20LangChain.md) @张友东 - - [x] [Index-1.9B-Chat WebDemo 部署](./bilibili_Index-1.9B/03-Index-1.9B-chat%20WebDemo部署.md) @九月 - - [x] [Index-1.9B-Chat Lora 微调](./bilibili_Index-1.9B/04-Index-1.9B-Chat%20Lora%20微调.md) @姜舒凡 + - [x] [Index-1.9B-Chat FastApi 部署调用](./models/bilibili_Index-1.9B/01-Index-1.9B-chat%20FastApi%20部署调用.md) @邓恺俊 + - [x] [Index-1.9B-Chat langchain 接入](./models/bilibili_Index-1.9B/02-Index-1.9B-Chat%20接入%20LangChain.md) @张友东 + - [x] [Index-1.9B-Chat WebDemo 部署](./models/bilibili_Index-1.9B/03-Index-1.9B-chat%20WebDemo部署.md) @九月 + - [x] [Index-1.9B-Chat Lora 微调](./models/bilibili_Index-1.9B/04-Index-1.9B-Chat%20Lora%20微调.md) @姜舒凡 - [Qwen2](https://github.com/QwenLM/Qwen2) - - [x] [Qwen2-7B-Instruct FastApi 部署调用](./Qwen2/01-Qwen2-7B-Instruct%20FastApi%20部署调用.md) @康婧淇 - - [x] [Qwen2-7B-Instruct langchain 接入](./Qwen2/02-Qwen2-7B-Instruct%20Langchain%20接入.md) @不要葱姜蒜 - - [x] [Qwen2-7B-Instruct WebDemo 部署](./Qwen2/03-Qwen2-7B-Instruct%20WebDemo部署.md) @三水 - - [x] [Qwen2-7B-Instruct vLLM 部署调用](./Qwen2/04-Qwen2-7B-Instruct%20vLLM%20部署调用.md) @姜舒凡 - - [x] [Qwen2-7B-Instruct Lora 微调](./Qwen2/05-Qwen2-7B-Instruct%20Lora%20微调.md) @散步 + - [x] [Qwen2-7B-Instruct FastApi 部署调用](./models/Qwen2/01-Qwen2-7B-Instruct%20FastApi%20部署调用.md) @康婧淇 + - [x] [Qwen2-7B-Instruct langchain 接入](./models/Qwen2/02-Qwen2-7B-Instruct%20Langchain%20接入.md) @不要葱姜蒜 + - [x] [Qwen2-7B-Instruct WebDemo 部署](./models/Qwen2/03-Qwen2-7B-Instruct%20WebDemo部署.md) @三水 + - [x] [Qwen2-7B-Instruct vLLM 部署调用](./models/Qwen2/04-Qwen2-7B-Instruct%20vLLM%20部署调用.md) @姜舒凡 + - [x] [Qwen2-7B-Instruct Lora 微调](./models/Qwen2/05-Qwen2-7B-Instruct%20Lora%20微调.md) @散步 - [GLM-4](https://github.com/THUDM/GLM-4.git) - - [x] [GLM-4-9B-chat FastApi 部署调用](./GLM-4/01-GLM-4-9B-chat%20FastApi%20部署调用.md) @张友东 - - [x] [GLM-4-9B-chat langchain 接入](./GLM-4/02-GLM-4-9B-chat%20langchain%20接入.md) @谭逸珂 - - [x] [GLM-4-9B-chat WebDemo 部署](./GLM-4/03-GLM-4-9B-Chat%20WebDemo.md) @何至轩 - - [x] [GLM-4-9B-chat vLLM 部署](./GLM-4/04-GLM-4-9B-Chat%20vLLM%20部署调用.md) @王熠明 - - [x] [GLM-4-9B-chat Lora 微调](./GLM-4/05-GLM-4-9B-chat%20Lora%20微调.md) @肖鸿儒 + - [x] [GLM-4-9B-chat FastApi 部署调用](./models/GLM-4/01-GLM-4-9B-chat%20FastApi%20部署调用.md) @张友东 + - [x] [GLM-4-9B-chat langchain 接入](./models/GLM-4/02-GLM-4-9B-chat%20langchain%20接入.md) @谭逸珂 + - [x] [GLM-4-9B-chat WebDemo 部署](./models/GLM-4/03-GLM-4-9B-Chat%20WebDemo.md) @何至轩 + - [x] [GLM-4-9B-chat vLLM 部署](./models/GLM-4/04-GLM-4-9B-Chat%20vLLM%20部署调用.md) @王熠明 + - [x] [GLM-4-9B-chat Lora 微调](./models/GLM-4/05-GLM-4-9B-chat%20Lora%20微调.md) @肖鸿儒 - [Qwen 1.5](https://github.com/QwenLM/Qwen1.5.git) - - [x] [Qwen1.5-7B-chat FastApi 部署调用](./Qwen1.5/01-Qwen1.5-7B-Chat%20FastApi%20部署调用.md) @颜鑫 - - [x] [Qwen1.5-7B-chat langchain 接入](./Qwen1.5/02-Qwen1.5-7B-Chat%20接入langchain搭建知识库助手.md) @颜鑫 - - [x] [Qwen1.5-7B-chat WebDemo 部署](./Qwen1.5/03-Qwen1.5-7B-Chat%20WebDemo.md) @颜鑫 - - [x] [Qwen1.5-7B-chat Lora 微调](./Qwen1.5/04-Qwen1.5-7B-chat%20Lora%20微调.md) @不要葱姜蒜 - - [x] [Qwen1.5-72B-chat-GPTQ-Int4 部署环境](./Qwen1.5/05-Qwen1.5-7B-Chat-GPTQ-Int4%20%20WebDemo.md) @byx020119 - - [x] [Qwen1.5-MoE-chat Transformers 部署调用](./Qwen1.5/06-Qwen1.5-MoE-A2.7B.md) @丁悦 - - [x] [Qwen1.5-7B-chat vLLM推理部署](./Qwen1.5/07-Qwen1.5-7B-Chat%20vLLM%20推理部署调用.md) @高立业 - - [x] [Qwen1.5-7B-chat Lora 微调 接入SwanLab实验管理平台](./Qwen1.5/08-Qwen1.5-7B-chat%20LoRA微调接入实验管理.md) @黄柏特 + - [x] [Qwen1.5-7B-chat FastApi 部署调用](./models/Qwen1.5/01-Qwen1.5-7B-Chat%20FastApi%20部署调用.md) @颜鑫 + - [x] [Qwen1.5-7B-chat langchain 接入](./models/Qwen1.5/02-Qwen1.5-7B-Chat%20接入langchain搭建知识库助手.md) @颜鑫 + - [x] [Qwen1.5-7B-chat WebDemo 部署](./models/Qwen1.5/03-Qwen1.5-7B-Chat%20WebDemo.md) @颜鑫 + - [x] [Qwen1.5-7B-chat Lora 微调](./models/Qwen1.5/04-Qwen1.5-7B-chat%20Lora%20微调.md) @不要葱姜蒜 + - [x] [Qwen1.5-72B-chat-GPTQ-Int4 部署环境](./models/Qwen1.5/05-Qwen1.5-7B-Chat-GPTQ-Int4%20%20WebDemo.md) @byx020119 + - [x] [Qwen1.5-MoE-chat Transformers 部署调用](./models/Qwen1.5/06-Qwen1.5-MoE-A2.7B.md) @丁悦 + - [x] [Qwen1.5-7B-chat vLLM推理部署](./models/Qwen1.5/07-Qwen1.5-7B-Chat%20vLLM%20推理部署调用.md) @高立业 + - [x] [Qwen1.5-7B-chat Lora 微调 接入SwanLab实验管理平台](./models/Qwen1.5/08-Qwen1.5-7B-chat%20LoRA微调接入实验管理.md) @黄柏特 - [谷歌-Gemma](https://huggingface.co/google/gemma-7b-it) - - [x] [gemma-2b-it FastApi 部署调用 ](./Gemma/01-Gemma-2B-Instruct%20FastApi%20部署调用.md) @东东 - - [x] [gemma-2b-it langchain 接入 ](./Gemma/02-Gemma-2B-Instruct%20langchain%20接入.md) @东东 - - [x] [gemma-2b-it WebDemo 部署 ](./Gemma/03-Gemma-2B-Instruct%20WebDemo%20部署.md) @东东 - - [x] [gemma-2b-it Peft Lora 微调 ](./Gemma/04-Gemma-2B-Instruct%20Lora微调.md) @东东 + - [x] [gemma-2b-it FastApi 部署调用 ](./models/Gemma/01-Gemma-2B-Instruct%20FastApi%20部署调用.md) @东东 + - [x] [gemma-2b-it langchain 接入 ](./models/Gemma/02-Gemma-2B-Instruct%20langchain%20接入.md) @东东 + - [x] [gemma-2b-it WebDemo 部署 ](./models/Gemma/03-Gemma-2B-Instruct%20WebDemo%20部署.md) @东东 + - [x] [gemma-2b-it Peft Lora 微调 ](./models/Gemma/04-Gemma-2B-Instruct%20Lora微调.md) @东东 - [phi-3](https://huggingface.co/microsoft/Phi-3-mini-4k-instruct) - - [x] [Phi-3-mini-4k-instruct FastApi 部署调用](./phi-3/01-Phi-3-mini-4k-instruct%20FastApi%20部署调用.md) @郑皓桦 - - [x] [Phi-3-mini-4k-instruct langchain 接入](./phi-3/02-Phi-3-mini-4k-instruct%20langchain%20接入.md) @郑皓桦 - - [x] [Phi-3-mini-4k-instruct WebDemo 部署](./phi-3/03-Phi-3-mini-4k-instruct%20WebDemo部署.md) @丁悦 - - [x] [Phi-3-mini-4k-instruct Lora 微调](./phi-3/04-Phi-3-mini-4k-Instruct%20Lora%20微调.md) @丁悦 + - [x] [Phi-3-mini-4k-instruct FastApi 部署调用](./models/phi-3/01-Phi-3-mini-4k-instruct%20FastApi%20部署调用.md) @郑皓桦 + - [x] [Phi-3-mini-4k-instruct langchain 接入](./models/phi-3/02-Phi-3-mini-4k-instruct%20langchain%20接入.md) @郑皓桦 + - [x] [Phi-3-mini-4k-instruct WebDemo 部署](./models/phi-3/03-Phi-3-mini-4k-instruct%20WebDemo部署.md) @丁悦 + - [x] [Phi-3-mini-4k-instruct Lora 微调](./models/phi-3/04-Phi-3-mini-4k-Instruct%20Lora%20微调.md) @丁悦 - [CharacterGLM-6B](https://github.com/thu-coai/CharacterGLM-6B) - - [x] [CharacterGLM-6B Transformers 部署调用](./CharacterGLM/01-CharacterGLM-6B%20Transformer部署调用.md) @孙健壮 - - [x] [CharacterGLM-6B FastApi 部署调用](./CharacterGLM/02-CharacterGLM-6B%20FastApi部署调用.md) @孙健壮 - - [x] [CharacterGLM-6B webdemo 部署](./CharacterGLM/03-CharacterGLM-6B-chat.md) @孙健壮 - - [x] [CharacterGLM-6B Lora 微调](./CharacterGLM/04-CharacterGLM-6B%20Lora微调.md) @孙健壮 + - [x] [CharacterGLM-6B Transformers 部署调用](./models/CharacterGLM/01-CharacterGLM-6B%20Transformer部署调用.md) @孙健壮 + - [x] [CharacterGLM-6B FastApi 部署调用](./models/CharacterGLM/02-CharacterGLM-6B%20FastApi部署调用.md) @孙健壮 + - [x] [CharacterGLM-6B webdemo 部署](./models/CharacterGLM/03-CharacterGLM-6B-chat.md) @孙健壮 + - [x] [CharacterGLM-6B Lora 微调](./models/CharacterGLM/04-CharacterGLM-6B%20Lora微调.md) @孙健壮 - [LLaMA3-8B-Instruct](https://github.com/meta-llama/llama3.git) - - [x] [LLaMA3-8B-Instruct FastApi 部署调用](./LLaMA3/01-LLaMA3-8B-Instruct%20FastApi%20部署调用.md) @高立业 - - [X] [LLaMA3-8B-Instruct langchain 接入](./LLaMA3/02-LLaMA3-8B-Instruct%20langchain%20接入.md) @不要葱姜蒜 - - [x] [LLaMA3-8B-Instruct WebDemo 部署](./LLaMA3/03-LLaMA3-8B-Instruct%20WebDemo%20部署.md) @不要葱姜蒜 - - [x] [LLaMA3-8B-Instruct Lora 微调](./LLaMA3/04-LLaMA3-8B-Instruct%20Lora%20微调.md) @高立业 + - [x] [LLaMA3-8B-Instruct FastApi 部署调用](./models/LLaMA3/01-LLaMA3-8B-Instruct%20FastApi%20部署调用.md) @高立业 + - [X] [LLaMA3-8B-Instruct langchain 接入](./models/LLaMA3/02-LLaMA3-8B-Instruct%20langchain%20接入.md) @不要葱姜蒜 + - [x] [LLaMA3-8B-Instruct WebDemo 部署](./models/LLaMA3/03-LLaMA3-8B-Instruct%20WebDemo%20部署.md) @不要葱姜蒜 + - [x] [LLaMA3-8B-Instruct Lora 微调](./models/LLaMA3/04-LLaMA3-8B-Instruct%20Lora%20微调.md) @高立业 - [XVERSE-7B-Chat](https://modelscope.cn/models/xverse/XVERSE-7B-Chat/summary) - - [x] [XVERSE-7B-Chat transformers 部署调用](./XVERSE/01-XVERSE-7B-chat%20Transformers推理.md) @郭志航 - - [x] [XVERSE-7B-Chat FastApi 部署调用](./XVERSE/02-XVERSE-7B-chat%20FastAPI部署.md) @郭志航 - - [x] [XVERSE-7B-Chat langchain 接入](./XVERSE/03-XVERSE-7B-chat%20langchain%20接入.md) @郭志航 - - [x] [XVERSE-7B-Chat WebDemo 部署](./XVERSE/04-XVERSE-7B-chat%20WebDemo%20部署.md) @郭志航 - - [x] [XVERSE-7B-Chat Lora 微调](./XVERSE/05-XVERSE-7B-Chat%20Lora%20微调.md) @郭志航 + - [x] [XVERSE-7B-Chat transformers 部署调用](./models/XVERSE/01-XVERSE-7B-chat%20Transformers推理.md) @郭志航 + - [x] [XVERSE-7B-Chat FastApi 部署调用](./models/XVERSE/02-XVERSE-7B-chat%20FastAPI部署.md) @郭志航 + - [x] [XVERSE-7B-Chat langchain 接入](./models/XVERSE/03-XVERSE-7B-chat%20langchain%20接入.md) @郭志航 + - [x] [XVERSE-7B-Chat WebDemo 部署](./models/XVERSE/04-XVERSE-7B-chat%20WebDemo%20部署.md) @郭志航 + - [x] [XVERSE-7B-Chat Lora 微调](./models/XVERSE/05-XVERSE-7B-Chat%20Lora%20微调.md) @郭志航 - [TransNormerLLM](https://github.com/OpenNLPLab/TransnormerLLM.git) - - [X] [TransNormerLLM-7B-Chat FastApi 部署调用](./TransNormer/01-TransNormer-7B%20FastApi%20部署调用.md) @王茂霖 - - [X] [TransNormerLLM-7B-Chat langchain 接入](./TransNormer/02-TransNormer-7B%20接入langchain搭建知识库助手.md) @王茂霖 - - [X] [TransNormerLLM-7B-Chat WebDemo 部署](./TransNormer/03-TransNormer-7B%20WebDemo.md) @王茂霖 - - [x] [TransNormerLLM-7B-Chat Lora 微调](./TransNormer/04-TrasnNormer-7B%20Lora%20微调.md) @王茂霖 + - [X] [TransNormerLLM-7B-Chat FastApi 部署调用](./models/TransNormer/01-TransNormer-7B%20FastApi%20部署调用.md) @王茂霖 + - [X] [TransNormerLLM-7B-Chat langchain 接入](./models/TransNormer/02-TransNormer-7B%20接入langchain搭建知识库助手.md) @王茂霖 + - [X] [TransNormerLLM-7B-Chat WebDemo 部署](./models/TransNormer/03-TransNormer-7B%20WebDemo.md) @王茂霖 + - [x] [TransNormerLLM-7B-Chat Lora 微调](./models/TransNormer/04-TrasnNormer-7B%20Lora%20微调.md) @王茂霖 - [BlueLM Vivo 蓝心大模型](https://github.com/vivo-ai-lab/BlueLM.git) - - [x] [BlueLM-7B-Chat FatApi 部署调用](./BlueLM/01-BlueLM-7B-Chat%20FastApi%20部署.md) @郭志航 - - [x] [BlueLM-7B-Chat langchain 接入](./BlueLM/02-BlueLM-7B-Chat%20langchain%20接入.md) @郭志航 - - [x] [BlueLM-7B-Chat WebDemo 部署](./BlueLM/03-BlueLM-7B-Chat%20WebDemo%20部署.md) @郭志航 - - [x] [BlueLM-7B-Chat Lora 微调](./BlueLM/04-BlueLM-7B-Chat%20Lora%20微调.md) @郭志航 + - [x] [BlueLM-7B-Chat FatApi 部署调用](./models/BlueLM/01-BlueLM-7B-Chat%20FastApi%20部署.md) @郭志航 + - [x] [BlueLM-7B-Chat langchain 接入](./models/BlueLM/02-BlueLM-7B-Chat%20langchain%20接入.md) @郭志航 + - [x] [BlueLM-7B-Chat WebDemo 部署](./models/BlueLM/03-BlueLM-7B-Chat%20WebDemo%20部署.md) @郭志航 + - [x] [BlueLM-7B-Chat Lora 微调](./models/BlueLM/04-BlueLM-7B-Chat%20Lora%20微调.md) @郭志航 - [InternLM2](https://github.com/InternLM/InternLM) - - [x] [InternLM2-7B-chat FastApi 部署调用](./InternLM2/01-InternLM2-7B-chat%20FastAPI部署.md) @不要葱姜蒜 - - [x] [InternLM2-7B-chat langchain 接入](./InternLM2/02-InternLM2-7B-chat%20langchain%20接入.md) @不要葱姜蒜 - - [x] [InternLM2-7B-chat WebDemo 部署](./InternLM2/03-InternLM2-7B-chat%20WebDemo%20部署.md) @郑皓桦 - - [x] [InternLM2-7B-chat Xtuner Qlora 微调](./InternLM2/04-InternLM2-7B-chat%20Xtuner%20Qlora%20微调.md) @郑皓桦 + - [x] [InternLM2-7B-chat FastApi 部署调用](./models/InternLM2/01-InternLM2-7B-chat%20FastAPI部署.md) @不要葱姜蒜 + - [x] [InternLM2-7B-chat langchain 接入](./models/InternLM2/02-InternLM2-7B-chat%20langchain%20接入.md) @不要葱姜蒜 + - [x] [InternLM2-7B-chat WebDemo 部署](./models/InternLM2/03-InternLM2-7B-chat%20WebDemo%20部署.md) @郑皓桦 + - [x] [InternLM2-7B-chat Xtuner Qlora 微调](./models/InternLM2/04-InternLM2-7B-chat%20Xtuner%20Qlora%20微调.md) @郑皓桦 - [DeepSeek 深度求索](https://github.com/deepseek-ai/DeepSeek-LLM) - - [x] [DeepSeek-7B-chat FastApi 部署调用](./DeepSeek/01-DeepSeek-7B-chat%20FastApi.md) @不要葱姜蒜 - - [x] [DeepSeek-7B-chat langchain 接入](./DeepSeek/02-DeepSeek-7B-chat%20langchain.md) @不要葱姜蒜 - - [x] [DeepSeek-7B-chat WebDemo](./DeepSeek/03-DeepSeek-7B-chat%20WebDemo.md) @不要葱姜蒜 - - [x] [DeepSeek-7B-chat Lora 微调](./DeepSeek/04-DeepSeek-7B-chat%20Lora%20微调.md) @不要葱姜蒜 - - [x] [DeepSeek-7B-chat 4bits量化 Qlora 微调](./DeepSeek/05-DeepSeek-7B-chat%204bits量化%20Qlora%20微调.md) @不要葱姜蒜 - - [x] [DeepSeek-MoE-16b-chat Transformers 部署调用](./DeepSeek/06-DeepSeek-MoE-16b-chat%20Transformer部署调用.md) @Kailigithub - - [x] [DeepSeek-MoE-16b-chat FastApi 部署调用](./DeepSeek/06-DeepSeek-MoE-16b-chat%20FastApi.md) @Kailigithub - - [x] [DeepSeek-coder-6.7b finetune colab](./DeepSeek/07-deepseek_fine_tune.ipynb) @Swiftie - - [x] [Deepseek-coder-6.7b webdemo colab](./DeepSeek/08-deepseek_web_demo.ipynb) @Swiftie + - [x] [DeepSeek-7B-chat FastApi 部署调用](./models/DeepSeek/01-DeepSeek-7B-chat%20FastApi.md) @不要葱姜蒜 + - [x] [DeepSeek-7B-chat langchain 接入](./models/DeepSeek/02-DeepSeek-7B-chat%20langchain.md) @不要葱姜蒜 + - [x] [DeepSeek-7B-chat WebDemo](./models/DeepSeek/03-DeepSeek-7B-chat%20WebDemo.md) @不要葱姜蒜 + - [x] [DeepSeek-7B-chat Lora 微调](./models/DeepSeek/04-DeepSeek-7B-chat%20Lora%20微调.md) @不要葱姜蒜 + - [x] [DeepSeek-7B-chat 4bits量化 Qlora 微调](./models/DeepSeek/05-DeepSeek-7B-chat%204bits量化%20Qlora%20微调.md) @不要葱姜蒜 + - [x] [DeepSeek-MoE-16b-chat Transformers 部署调用](./models/DeepSeek/06-DeepSeek-MoE-16b-chat%20Transformer部署调用.md) @Kailigithub + - [x] [DeepSeek-MoE-16b-chat FastApi 部署调用](./models/DeepSeek/06-DeepSeek-MoE-16b-chat%20FastApi.md) @Kailigithub + - [x] [DeepSeek-coder-6.7b finetune colab](./models/DeepSeek/07-deepseek_fine_tune.ipynb) @Swiftie + - [x] [Deepseek-coder-6.7b webdemo colab](./models/DeepSeek/08-deepseek_web_demo.ipynb) @Swiftie - [MiniCPM](https://github.com/OpenBMB/MiniCPM.git) - - [x] [MiniCPM-2B-chat transformers 部署调用](./MiniCPM/MiniCPM-2B-chat%20transformers%20部署调用.md) @Kailigithub - - [x] [MiniCPM-2B-chat FastApi 部署调用](./MiniCPM/MiniCPM-2B-chat%20FastApi%20部署调用.md) @Kailigithub - - [x] [MiniCPM-2B-chat langchain 接入](./MiniCPM/MiniCPM-2B-chat%20langchain接入.md) @不要葱姜蒜 - - [x] [MiniCPM-2B-chat webdemo 部署](./MiniCPM/MiniCPM-2B-chat%20WebDemo部署.md) @Kailigithub - - [x] [MiniCPM-2B-chat Lora && Full 微调](./MiniCPM/MiniCPM-2B-chat%20Lora%20&&%20Full%20微调.md) @不要葱姜蒜 + - [x] [MiniCPM-2B-chat transformers 部署调用](./models/MiniCPM/MiniCPM-2B-chat%20transformers%20部署调用.md) @Kailigithub + - [x] [MiniCPM-2B-chat FastApi 部署调用](./models/MiniCPM/MiniCPM-2B-chat%20FastApi%20部署调用.md) @Kailigithub + - [x] [MiniCPM-2B-chat langchain 接入](./models/MiniCPM/MiniCPM-2B-chat%20langchain接入.md) @不要葱姜蒜 + - [x] [MiniCPM-2B-chat webdemo 部署](./models/MiniCPM/MiniCPM-2B-chat%20WebDemo部署.md) @Kailigithub + - [x] [MiniCPM-2B-chat Lora && Full 微调](./models/MiniCPM/MiniCPM-2B-chat%20Lora%20&&%20Full%20微调.md) @不要葱姜蒜 - [Qwen-Audio](https://github.com/QwenLM/Qwen-Audio.git) - - [x] [Qwen-Audio FastApi 部署调用](./Qwen-Audio/01-Qwen-Audio-chat%20FastApi.md) @陈思州 - - [x] [Qwen-Audio WebDemo](./Qwen-Audio/02-Qwen-Audio-chat%20WebDemo.md) @陈思州 + - [x] [Qwen-Audio FastApi 部署调用](./models/Qwen-Audio/01-Qwen-Audio-chat%20FastApi.md) @陈思州 + - [x] [Qwen-Audio WebDemo](./models/Qwen-Audio/02-Qwen-Audio-chat%20WebDemo.md) @陈思州 - [Qwen](https://github.com/QwenLM/Qwen.git) - - [x] [Qwen-7B-chat Transformers 部署调用](./Qwen/01-Qwen-7B-Chat%20Transformers部署调用.md) @李娇娇 - - [x] [Qwen-7B-chat FastApi 部署调用](./Qwen/02-Qwen-7B-Chat%20FastApi%20部署调用.md) @李娇娇 - - [x] [Qwen-7B-chat WebDemo](./Qwen/03-Qwen-7B-Chat%20WebDemo.md) @李娇娇 - - [x] [Qwen-7B-chat Lora 微调](./Qwen/04-Qwen-7B-Chat%20Lora%20微调.md) @不要葱姜蒜 - - [x] [Qwen-7B-chat ptuning 微调](./Qwen/05-Qwen-7B-Chat%20Ptuning%20微调.md) @肖鸿儒 - - [x] [Qwen-7B-chat 全量微调](./Qwen/06-Qwen-7B-chat%20全量微调.md) @不要葱姜蒜 - - [x] [Qwen-7B-Chat 接入langchain搭建知识库助手](./Qwen/07-Qwen-7B-Chat%20接入langchain搭建知识库助手.md) @李娇娇 - - [x] [Qwen-7B-chat 低精度训练](./Qwen/08-Qwen-7B-Chat%20Lora%20低精度微调.md) @肖鸿儒 - - [x] [Qwen-1_8B-chat CPU 部署](./Qwen/09-Qwen-1_8B-chat%20CPU%20部署%20.md) @散步 + - [x] [Qwen-7B-chat Transformers 部署调用](./models/Qwen/01-Qwen-7B-Chat%20Transformers部署调用.md) @李娇娇 + - [x] [Qwen-7B-chat FastApi 部署调用](./models/Qwen/02-Qwen-7B-Chat%20FastApi%20部署调用.md) @李娇娇 + - [x] [Qwen-7B-chat WebDemo](./models/Qwen/03-Qwen-7B-Chat%20WebDemo.md) @李娇娇 + - [x] [Qwen-7B-chat Lora 微调](./models/Qwen/04-Qwen-7B-Chat%20Lora%20微调.md) @不要葱姜蒜 + - [x] [Qwen-7B-chat ptuning 微调](./models/Qwen/05-Qwen-7B-Chat%20Ptuning%20微调.md) @肖鸿儒 + - [x] [Qwen-7B-chat 全量微调](./models/Qwen/06-Qwen-7B-chat%20全量微调.md) @不要葱姜蒜 + - [x] [Qwen-7B-Chat 接入langchain搭建知识库助手](./models/Qwen/07-Qwen-7B-Chat%20接入langchain搭建知识库助手.md) @李娇娇 + - [x] [Qwen-7B-chat 低精度训练](./models/Qwen/08-Qwen-7B-Chat%20Lora%20低精度微调.md) @肖鸿儒 + - [x] [Qwen-1_8B-chat CPU 部署](./models/Qwen/09-Qwen-1_8B-chat%20CPU%20部署%20.md) @散步 - [Yi 零一万物](https://github.com/01-ai/Yi.git) - - [x] [Yi-6B-chat FastApi 部署调用](./Yi/01-Yi-6B-Chat%20FastApi%20部署调用.md) @李柯辰 - - [x] [Yi-6B-chat langchain接入](./Yi/02-Yi-6B-Chat%20接入langchain搭建知识库助手.md) @李柯辰 - - [x] [Yi-6B-chat WebDemo](./Yi/03-Yi-6B-chat%20WebDemo.md) @肖鸿儒 - - [x] [Yi-6B-chat Lora 微调](./Yi/04-Yi-6B-Chat%20Lora%20微调.md) @李娇娇 + - [x] [Yi-6B-chat FastApi 部署调用](./models/Yi/01-Yi-6B-Chat%20FastApi%20部署调用.md) @李柯辰 + - [x] [Yi-6B-chat langchain接入](./models/Yi/02-Yi-6B-Chat%20接入langchain搭建知识库助手.md) @李柯辰 + - [x] [Yi-6B-chat WebDemo](./models/Yi/03-Yi-6B-chat%20WebDemo.md) @肖鸿儒 + - [x] [Yi-6B-chat Lora 微调](./models/Yi/04-Yi-6B-Chat%20Lora%20微调.md) @李娇娇 - [Baichuan 百川智能](https://www.baichuan-ai.com/home) - [x] [Baichuan2-7B-chat FastApi 部署调用](./BaiChuan/01-Baichuan2-7B-chat%2BFastApi%2B%E9%83%A8%E7%BD%B2%E8%B0%83%E7%94%A8.md) @惠佳豪 - - [x] [Baichuan2-7B-chat WebDemo](./BaiChuan/02-Baichuan-7B-chat%2BWebDemo.md) @惠佳豪 - - [x] [Baichuan2-7B-chat 接入 LangChain 框架](./BaiChuan/03-Baichuan2-7B-chat%E6%8E%A5%E5%85%A5LangChain%E6%A1%86%E6%9E%B6.md) @惠佳豪 - - [x] [Baichuan2-7B-chat Lora 微调](./BaiChuan/04-Baichuan2-7B-chat%2Blora%2B%E5%BE%AE%E8%B0%83.md) @惠佳豪 + - [x] [Baichuan2-7B-chat WebDemo](./models/BaiChuan/02-Baichuan-7B-chat%2BWebDemo.md) @惠佳豪 + - [x] [Baichuan2-7B-chat 接入 LangChain 框架](./models/BaiChuan/03-Baichuan2-7B-chat%E6%8E%A5%E5%85%A5LangChain%E6%A1%86%E6%9E%B6.md) @惠佳豪 + - [x] [Baichuan2-7B-chat Lora 微调](./models/BaiChuan/04-Baichuan2-7B-chat%2Blora%2B%E5%BE%AE%E8%B0%83.md) @惠佳豪 - [InternLM](https://github.com/InternLM/InternLM.git) - - [x] [InternLM-Chat-7B Transformers 部署调用](./InternLM/01-InternLM-Chat-7B%20Transformers%20部署调用.md) @小罗 - - [x] [InternLM-Chat-7B FastApi 部署调用](InternLM/02-internLM-Chat-7B%20FastApi.md) @不要葱姜蒜 - - [x] [InternLM-Chat-7B WebDemo](InternLM/03-InternLM-Chat-7B.md) @不要葱姜蒜 - - [x] [Lagent+InternLM-Chat-7B-V1.1 WebDemo](InternLM/04-Lagent+InternLM-Chat-7B-V1.1.md) @不要葱姜蒜 - - [x] [浦语灵笔图文理解&创作 WebDemo](InternLM/05-浦语灵笔图文理解&创作.md) @不要葱姜蒜 - - [x] [InternLM-Chat-7B 接入 LangChain 框架](InternLM/06-InternLM接入LangChain搭建知识库助手.md) @Logan Zou + - [x] [InternLM-Chat-7B Transformers 部署调用](./models/InternLM/01-InternLM-Chat-7B%20Transformers%20部署调用.md) @小罗 + - [x] [InternLM-Chat-7B FastApi 部署调用](./models/InternLM/02-internLM-Chat-7B%20FastApi.md) @不要葱姜蒜 + - [x] [InternLM-Chat-7B WebDemo](./models/InternLM/03-InternLM-Chat-7B.md) @不要葱姜蒜 + - [x] [Lagent+InternLM-Chat-7B-V1.1 WebDemo](./models/InternLM/04-Lagent+InternLM-Chat-7B-V1.1.md) @不要葱姜蒜 + - [x] [浦语灵笔图文理解&创作 WebDemo](./models/InternLM/05-浦语灵笔图文理解&创作.md) @不要葱姜蒜 + - [x] [InternLM-Chat-7B 接入 LangChain 框架](./models/InternLM/06-InternLM接入LangChain搭建知识库助手.md) @Logan Zou - [Atom (llama2)](https://hf-mirror.com/FlagAlpha/Atom-7B-Chat) - - [x] [Atom-7B-chat WebDemo](./Atom/01-Atom-7B-chat-WebDemo.md) @Kailigithub - - [x] [Atom-7B-chat Lora 微调](./Atom/02-Atom-7B-Chat%20Lora%20微调.md) @Logan Zou - - [x] [Atom-7B-Chat 接入langchain搭建知识库助手](./Atom/03-Atom-7B-Chat%20接入langchain搭建知识库助手.md) @陈思州 - - [x] [Atom-7B-chat 全量微调](./Atom/04-Atom-7B-chat%20全量微调.md) @Logan Zou + - [x] [Atom-7B-chat WebDemo](./models/Atom/01-Atom-7B-chat-WebDemo.md) @Kailigithub + - [x] [Atom-7B-chat Lora 微调](./models/Atom/02-Atom-7B-Chat%20Lora%20微调.md) @Logan Zou + - [x] [Atom-7B-Chat 接入langchain搭建知识库助手](./models/Atom/03-Atom-7B-Chat%20接入langchain搭建知识库助手.md) @陈思州 + - [x] [Atom-7B-chat 全量微调](./models/Atom/04-Atom-7B-chat%20全量微调.md) @Logan Zou - [ChatGLM3](https://github.com/THUDM/ChatGLM3.git) - - [x] [ChatGLM3-6B Transformers 部署调用](./ChatGLM/01-ChatGLM3-6B%20Transformer部署调用.md) @丁悦 - - [x] [ChatGLM3-6B FastApi 部署调用](./ChatGLM/02-ChatGLM3-6B%20FastApi部署调用.md) @丁悦 - - [x] [ChatGLM3-6B chat WebDemo](ChatGLM/03-ChatGLM3-6B-chat.md) @不要葱姜蒜 - - [x] [ChatGLM3-6B Code Interpreter WebDemo](ChatGLM/04-ChatGLM3-6B-Code-Interpreter.md) @不要葱姜蒜 - - [x] [ChatGLM3-6B 接入 LangChain 框架](ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手.md) @Logan Zou - - [x] [ChatGLM3-6B Lora 微调](ChatGLM/06-ChatGLM3-6B-Lora微调.md) @肖鸿儒 + - [x] [ChatGLM3-6B Transformers 部署调用](./models/ChatGLM/01-ChatGLM3-6B%20Transformer部署调用.md) @丁悦 + - [x] [ChatGLM3-6B FastApi 部署调用](./models/ChatGLM/02-ChatGLM3-6B%20FastApi部署调用.md) @丁悦 + - [x] [ChatGLM3-6B chat WebDemo](./models/ChatGLM/03-ChatGLM3-6B-chat.md) @不要葱姜蒜 + - [x] [ChatGLM3-6B Code Interpreter WebDemo](./models/ChatGLM/04-ChatGLM3-6B-Code-Interpreter.md) @不要葱姜蒜 + - [x] [ChatGLM3-6B 接入 LangChain 框架](./models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手.md) @Logan Zou + - [x] [ChatGLM3-6B Lora 微调](./models/ChatGLM/06-ChatGLM3-6B-Lora微调.md) @肖鸿儒 ### 通用环境配置 -- [x] [pip、conda 换源](./General-Setting/01-pip、conda换源.md) @不要葱姜蒜 -- [x] [AutoDL 开放端口](./General-Setting/02-AutoDL开放端口.md) @不要葱姜蒜 +- [x] [pip、conda 换源](./models/General-Setting/01-pip、conda换源.md) @不要葱姜蒜 +- [x] [AutoDL 开放端口](./models/General-Setting/02-AutoDL开放端口.md) @不要葱姜蒜 - 模型下载 - - [x] [hugging face](./General-Setting/03-模型下载.md) @不要葱姜蒜 + - [x] [hugging face](./models/General-Setting/03-模型下载.md) @不要葱姜蒜 - [x] [hugging face](./General-Setting/03-模型下载.md) 镜像下载 @不要葱姜蒜 - - [x] [modelscope](./General-Setting/03-模型下载.md) @不要葱姜蒜 - - [x] [git-lfs](./General-Setting/03-模型下载.md) @不要葱姜蒜 - - [x] [Openxlab](./General-Setting/03-模型下载.md) + - [x] [modelscope](./models/General-Setting/03-模型下载.md) @不要葱姜蒜 + - [x] [git-lfs](./models/General-Setting/03-模型下载.md) @不要葱姜蒜 + - [x] [Openxlab](./models/General-Setting/03-模型下载.md) - Issue && PR - - [x] [Issue 提交](./General-Setting/04-Issue&PR&update.md) @肖鸿儒 - - [x] [PR 提交](./General-Setting/04-Issue&PR&update.md) @肖鸿儒 - - [x] [fork更新](./General-Setting/04-Issue&PR&update.md) @肖鸿儒 + - [x] [Issue 提交](./models/General-Setting/04-Issue&PR&update.md) @肖鸿儒 + - [x] [PR 提交](./models/General-Setting/04-Issue&PR&update.md) @肖鸿儒 + - [x] [fork更新](./models/General-Setting/04-Issue&PR&update.md) @肖鸿儒 ## 致谢 diff --git a/examples/readme.md b/examples/readme.md new file mode 100644 index 0000000..45dd6f2 --- /dev/null +++ b/examples/readme.md @@ -0,0 +1,13 @@ +# self-llm Examples + +在学习者完成了基础部分的学习之后,我们将会提供一些例子来帮助学习者更好的理解和掌握大模型应用开发。我们会以应用类型为向导,提供一些优秀的大模型应用项目案例 Demo ,使得学习者完成我们的 examples 后,能够更好的理解和掌握大模型应用开发的技术要点和掌握大模型应用二次开发或单独开发的能力。 + +## Examples 目录 + +- 角色扮演 + - [ ] Chat-嬛嬛 + - [ ] Chat-悟空 +- 办公效率 + - [ ] 搭建 RAG 对话系统 +- 学习教育 + - [ ] ChatTest \ No newline at end of file diff --git a/Atom/01-Atom-7B-chat-WebDemo.md b/models/Atom/01-Atom-7B-chat-WebDemo.md similarity index 100% rename from Atom/01-Atom-7B-chat-WebDemo.md rename to models/Atom/01-Atom-7B-chat-WebDemo.md diff --git a/Atom/02-Atom-7B-Chat Lora 微调.md b/models/Atom/02-Atom-7B-Chat Lora 微调.md similarity index 100% rename from Atom/02-Atom-7B-Chat Lora 微调.md rename to models/Atom/02-Atom-7B-Chat Lora 微调.md diff --git a/Atom/02-Atom-7B-Chat-Lora/train.py b/models/Atom/02-Atom-7B-Chat-Lora/train.py similarity index 100% rename from Atom/02-Atom-7B-Chat-Lora/train.py rename to models/Atom/02-Atom-7B-Chat-Lora/train.py diff --git a/Atom/02-Atom-7B-Chat-Lora/train.sh b/models/Atom/02-Atom-7B-Chat-Lora/train.sh similarity index 100% rename from Atom/02-Atom-7B-Chat-Lora/train.sh rename to models/Atom/02-Atom-7B-Chat-Lora/train.sh diff --git a/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手.md b/models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手.md similarity index 100% rename from Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手.md rename to models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手.md diff --git a/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/LLM.py b/models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/LLM.py similarity index 100% rename from Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/LLM.py rename to models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/LLM.py diff --git a/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/creat_db.py b/models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/creat_db.py similarity index 100% rename from Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/creat_db.py rename to models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/creat_db.py diff --git a/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/readme.md b/models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/readme.md similarity index 100% rename from Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/readme.md rename to models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/readme.md diff --git a/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/run_gradio.py b/models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/run_gradio.py similarity index 100% rename from Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/run_gradio.py rename to models/Atom/03-Atom-7B-Chat 接入langchain搭建知识库助手/run_gradio.py diff --git a/Atom/04-Atom-7B-chat 全量微调.md b/models/Atom/04-Atom-7B-chat 全量微调.md similarity index 100% rename from Atom/04-Atom-7B-chat 全量微调.md rename to models/Atom/04-Atom-7B-chat 全量微调.md diff --git a/Atom/images/image-1.png b/models/Atom/images/image-1.png similarity index 100% rename from Atom/images/image-1.png rename to models/Atom/images/image-1.png diff --git a/Atom/images/image-2.png b/models/Atom/images/image-2.png similarity index 100% rename from Atom/images/image-2.png rename to models/Atom/images/image-2.png diff --git a/Atom/images/image-3.png b/models/Atom/images/image-3.png similarity index 100% rename from Atom/images/image-3.png rename to models/Atom/images/image-3.png diff --git a/Atom/images/image-4.png b/models/Atom/images/image-4.png similarity index 100% rename from Atom/images/image-4.png rename to models/Atom/images/image-4.png diff --git a/Atom/images/image-5.png b/models/Atom/images/image-5.png similarity index 100% rename from Atom/images/image-5.png rename to models/Atom/images/image-5.png diff --git a/Atom/images/image-6.png b/models/Atom/images/image-6.png similarity index 100% rename from Atom/images/image-6.png rename to models/Atom/images/image-6.png diff --git a/Atom/images/image-7.png b/models/Atom/images/image-7.png similarity index 100% rename from Atom/images/image-7.png rename to models/Atom/images/image-7.png diff --git a/Atom/images/image-8.png b/models/Atom/images/image-8.png similarity index 100% rename from Atom/images/image-8.png rename to models/Atom/images/image-8.png diff --git a/Atom/images/image-9.png b/models/Atom/images/image-9.png similarity index 100% rename from Atom/images/image-9.png rename to models/Atom/images/image-9.png diff --git a/BaiChuan/01-Baichuan2-7B-chat+FastApi+部署调用.md b/models/BaiChuan/01-Baichuan2-7B-chat+FastApi+部署调用.md similarity index 100% rename from BaiChuan/01-Baichuan2-7B-chat+FastApi+部署调用.md rename to models/BaiChuan/01-Baichuan2-7B-chat+FastApi+部署调用.md diff --git a/BaiChuan/02-Baichuan-7B-chat+WebDemo.md b/models/BaiChuan/02-Baichuan-7B-chat+WebDemo.md similarity index 100% rename from BaiChuan/02-Baichuan-7B-chat+WebDemo.md rename to models/BaiChuan/02-Baichuan-7B-chat+WebDemo.md diff --git a/BaiChuan/03-Baichuan2-7B-chat接入LangChain框架.md b/models/BaiChuan/03-Baichuan2-7B-chat接入LangChain框架.md similarity index 100% rename from BaiChuan/03-Baichuan2-7B-chat接入LangChain框架.md rename to models/BaiChuan/03-Baichuan2-7B-chat接入LangChain框架.md diff --git a/BaiChuan/04-Baichuan2-7B-chat Lora 微调.ipynb b/models/BaiChuan/04-Baichuan2-7B-chat Lora 微调.ipynb similarity index 100% rename from BaiChuan/04-Baichuan2-7B-chat Lora 微调.ipynb rename to models/BaiChuan/04-Baichuan2-7B-chat Lora 微调.ipynb diff --git a/BaiChuan/04-Baichuan2-7B-chat+lora+微调.md b/models/BaiChuan/04-Baichuan2-7B-chat+lora+微调.md similarity index 100% rename from BaiChuan/04-Baichuan2-7B-chat+lora+微调.md rename to models/BaiChuan/04-Baichuan2-7B-chat+lora+微调.md diff --git a/BaiChuan/images/image1.png b/models/BaiChuan/images/image1.png similarity index 100% rename from BaiChuan/images/image1.png rename to models/BaiChuan/images/image1.png diff --git a/BaiChuan/images/image10.png b/models/BaiChuan/images/image10.png similarity index 100% rename from BaiChuan/images/image10.png rename to models/BaiChuan/images/image10.png diff --git a/BaiChuan/images/image11.png b/models/BaiChuan/images/image11.png similarity index 100% rename from BaiChuan/images/image11.png rename to models/BaiChuan/images/image11.png diff --git a/BaiChuan/images/image12.png b/models/BaiChuan/images/image12.png similarity index 100% rename from BaiChuan/images/image12.png rename to models/BaiChuan/images/image12.png diff --git a/BaiChuan/images/image13.png b/models/BaiChuan/images/image13.png similarity index 100% rename from BaiChuan/images/image13.png rename to models/BaiChuan/images/image13.png diff --git a/BaiChuan/images/image14.png b/models/BaiChuan/images/image14.png similarity index 100% rename from BaiChuan/images/image14.png rename to models/BaiChuan/images/image14.png diff --git a/BaiChuan/images/image15.png b/models/BaiChuan/images/image15.png similarity index 100% rename from BaiChuan/images/image15.png rename to models/BaiChuan/images/image15.png diff --git a/BaiChuan/images/image16.png b/models/BaiChuan/images/image16.png similarity index 100% rename from BaiChuan/images/image16.png rename to models/BaiChuan/images/image16.png diff --git a/BaiChuan/images/image17.png b/models/BaiChuan/images/image17.png similarity index 100% rename from BaiChuan/images/image17.png rename to models/BaiChuan/images/image17.png diff --git a/BaiChuan/images/image18.png b/models/BaiChuan/images/image18.png similarity index 100% rename from BaiChuan/images/image18.png rename to models/BaiChuan/images/image18.png diff --git a/BaiChuan/images/image2.png b/models/BaiChuan/images/image2.png similarity index 100% rename from BaiChuan/images/image2.png rename to models/BaiChuan/images/image2.png diff --git a/BaiChuan/images/image20.png b/models/BaiChuan/images/image20.png similarity index 100% rename from BaiChuan/images/image20.png rename to models/BaiChuan/images/image20.png diff --git a/BaiChuan/images/image23.png b/models/BaiChuan/images/image23.png similarity index 100% rename from BaiChuan/images/image23.png rename to models/BaiChuan/images/image23.png diff --git a/BaiChuan/images/image25.png b/models/BaiChuan/images/image25.png similarity index 100% rename from BaiChuan/images/image25.png rename to models/BaiChuan/images/image25.png diff --git a/BaiChuan/images/image26.png b/models/BaiChuan/images/image26.png similarity index 100% rename from BaiChuan/images/image26.png rename to models/BaiChuan/images/image26.png diff --git a/BaiChuan/images/image27.png b/models/BaiChuan/images/image27.png similarity index 100% rename from BaiChuan/images/image27.png rename to models/BaiChuan/images/image27.png diff --git a/BaiChuan/images/image3.png b/models/BaiChuan/images/image3.png similarity index 100% rename from BaiChuan/images/image3.png rename to models/BaiChuan/images/image3.png diff --git a/BaiChuan/images/image4.png b/models/BaiChuan/images/image4.png similarity index 100% rename from BaiChuan/images/image4.png rename to models/BaiChuan/images/image4.png diff --git a/BaiChuan/images/image6.png b/models/BaiChuan/images/image6.png similarity index 100% rename from BaiChuan/images/image6.png rename to models/BaiChuan/images/image6.png diff --git a/BaiChuan/images/image7.png b/models/BaiChuan/images/image7.png similarity index 100% rename from BaiChuan/images/image7.png rename to models/BaiChuan/images/image7.png diff --git a/BaiChuan/images/image8.png b/models/BaiChuan/images/image8.png similarity index 100% rename from BaiChuan/images/image8.png rename to models/BaiChuan/images/image8.png diff --git a/BaiChuan/images/image9.png b/models/BaiChuan/images/image9.png similarity index 100% rename from BaiChuan/images/image9.png rename to models/BaiChuan/images/image9.png diff --git a/BlueLM/01-BlueLM-7B-Chat FastApi 部署.md b/models/BlueLM/01-BlueLM-7B-Chat FastApi 部署.md similarity index 100% rename from BlueLM/01-BlueLM-7B-Chat FastApi 部署.md rename to models/BlueLM/01-BlueLM-7B-Chat FastApi 部署.md diff --git a/BlueLM/02-BlueLM-7B-Chat langchain 接入.md b/models/BlueLM/02-BlueLM-7B-Chat langchain 接入.md similarity index 100% rename from BlueLM/02-BlueLM-7B-Chat langchain 接入.md rename to models/BlueLM/02-BlueLM-7B-Chat langchain 接入.md diff --git a/BlueLM/03-BlueLM-7B-Chat WebDemo 部署.md b/models/BlueLM/03-BlueLM-7B-Chat WebDemo 部署.md similarity index 100% rename from BlueLM/03-BlueLM-7B-Chat WebDemo 部署.md rename to models/BlueLM/03-BlueLM-7B-Chat WebDemo 部署.md diff --git a/BlueLM/04-BlueLM-7B-Chat Lora 微调.ipynb b/models/BlueLM/04-BlueLM-7B-Chat Lora 微调.ipynb similarity index 100% rename from BlueLM/04-BlueLM-7B-Chat Lora 微调.ipynb rename to models/BlueLM/04-BlueLM-7B-Chat Lora 微调.ipynb diff --git a/BlueLM/04-BlueLM-7B-Chat Lora 微调.md b/models/BlueLM/04-BlueLM-7B-Chat Lora 微调.md similarity index 100% rename from BlueLM/04-BlueLM-7B-Chat Lora 微调.md rename to models/BlueLM/04-BlueLM-7B-Chat Lora 微调.md diff --git a/BlueLM/04-BlueLM-7B-Chat Lora 微调.py b/models/BlueLM/04-BlueLM-7B-Chat Lora 微调.py similarity index 100% rename from BlueLM/04-BlueLM-7B-Chat Lora 微调.py rename to models/BlueLM/04-BlueLM-7B-Chat Lora 微调.py diff --git a/BlueLM/images/202403191628941.png b/models/BlueLM/images/202403191628941.png similarity index 100% rename from BlueLM/images/202403191628941.png rename to models/BlueLM/images/202403191628941.png diff --git a/BlueLM/images/202403191813385.png b/models/BlueLM/images/202403191813385.png similarity index 100% rename from BlueLM/images/202403191813385.png rename to models/BlueLM/images/202403191813385.png diff --git a/BlueLM/images/202403201210690.png b/models/BlueLM/images/202403201210690.png similarity index 100% rename from BlueLM/images/202403201210690.png rename to models/BlueLM/images/202403201210690.png diff --git a/BlueLM/images/202403201229542.png b/models/BlueLM/images/202403201229542.png similarity index 100% rename from BlueLM/images/202403201229542.png rename to models/BlueLM/images/202403201229542.png diff --git a/BlueLM/images/202403202153465.png b/models/BlueLM/images/202403202153465.png similarity index 100% rename from BlueLM/images/202403202153465.png rename to models/BlueLM/images/202403202153465.png diff --git a/CharacterGLM/01-CharacterGLM-6B Transformer部署调用.md b/models/CharacterGLM/01-CharacterGLM-6B Transformer部署调用.md similarity index 98% rename from CharacterGLM/01-CharacterGLM-6B Transformer部署调用.md rename to models/CharacterGLM/01-CharacterGLM-6B Transformer部署调用.md index 03735d8..97cd315 100644 --- a/CharacterGLM/01-CharacterGLM-6B Transformer部署调用.md +++ b/models/CharacterGLM/01-CharacterGLM-6B Transformer部署调用.md @@ -1,77 +1,77 @@ -# CharacterGLM-6B Transformers部署调用 - -## 环境准备 - -在autodl平台中租一个3090等24G显存的显卡机器,如下图所示镜像选择PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/fc4c6323-d338-4d66-a244-bbefe7da3746) - -接下来打开刚刚租用服务器的JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行demo。 - -pip换源和安装依赖包 - -```python -#升级pip -python -m pip install --upgrade pip -#更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install modelscope -pip install transformers -pip install sentencepiece -``` - -## 模型下载 - -使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 python /root/autodl-tmp/download.py执行下载,模型大小为 12 GB,下载模型大概需要 10~15 分钟 - -```python -import torch -from modelscope import snapshot_download, AutoModel, AutoTokenizer -import os -model_dir = snapshot_download('THUCoAI/CharacterGLM-6B', cache_dir='/root/autodl-tmp', revision='master') -``` - -## 代码准备 - -```python -from transformers import AutoTokenizer,AutoModelForCausalLM -import torch -# 使用模型下载到的本地路径以加载 -model_dir = '/root/autodl-tmp/THUCoAI/CharacterGLM-6B' -# 分词器的加载,本地加载,trust_remote_code=True设置允许从网络上下载模型权重和相关的代码 -tokenizer = AutoTokenizer.from_pretrained(model_dir, trust_remote_code=True) -# 模型加载,本地加载,使用AutoModelForCausalLM类 -model = AutoModelForCausalLM.from_pretrained(model_dir, trust_remote_code=True) -# 将模型移动到GPU上进行加速(如果有GPU的话) -device = torch.device("cuda" if torch.cuda.is_available() else "cpu") -model.to(device) -# 使用模型的评估模式来产生对话 -model.eval() -session_meta = {'user_info': '我是陆星辰,是一个男性,是一位知名导演,也是苏梦远的合作导演。我擅长拍摄音乐题材的电影。苏梦远对我的态度是尊敬的,并视我为良师益友。', 'bot_info': '苏梦远,本名苏远心,是一位当红的国内女歌手及演员。在参加选秀节目后,凭借独特的嗓音及出众的舞台魅力迅速成名,进入娱乐圈。她外表美丽动人,但真正的魅力在于她的才华和勤奋。苏梦远是音乐学院毕业的优秀生,善于创作,拥有多首热门原创歌曲。除了音乐方面的成就,她还热衷于慈善事业,积极参加公益活动,用实际行动传递正能量。在工作中,她对待工作非常敬业,拍戏时总是全身心投入角色,赢得了业内人士的赞誉和粉丝的喜爱。虽然在娱乐圈,但她始终保持低调、谦逊的态度,深得同行尊重。在表达时,苏梦远喜欢使用“我们”和“一起”,强调团队精神。', 'bot_name': '苏梦远', 'user_name': '陆星辰'} -# 第一轮对话 -response, history = model.chat(tokenizer, session_meta,"你好呀,小苏", history=[]) -print(response) -# 第二轮对话 -response, history = model.chat(tokenizer, session_meta,"最近对音乐有什么新的想法吗", history=history) -print(response) -# 第三轮对话 -response, history = model.chat(tokenizer,session_meta, "那我们商量一下下一部音乐电影的拍摄,好嘛?", history=history) -print(response) -``` - -## 部署 - -在终端输入以下命令运行trans.py,即实现CharacterGLM-6B的Transformers部署调用 - -```python -cd /root/autodl-tmp -python trans.py -``` - -观察命令行中loading checkpoint表示模型正在加载,等待模型加载完成产生对话,如下图所示 - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/f9d65275-fa89-4039-95c5-7cc0615753e2) - +# CharacterGLM-6B Transformers部署调用 + +## 环境准备 + +在autodl平台中租一个3090等24G显存的显卡机器,如下图所示镜像选择PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/fc4c6323-d338-4d66-a244-bbefe7da3746) + +接下来打开刚刚租用服务器的JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行demo。 + +pip换源和安装依赖包 + +```python +#升级pip +python -m pip install --upgrade pip +#更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install modelscope +pip install transformers +pip install sentencepiece +``` + +## 模型下载 + +使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 python /root/autodl-tmp/download.py执行下载,模型大小为 12 GB,下载模型大概需要 10~15 分钟 + +```python +import torch +from modelscope import snapshot_download, AutoModel, AutoTokenizer +import os +model_dir = snapshot_download('THUCoAI/CharacterGLM-6B', cache_dir='/root/autodl-tmp', revision='master') +``` + +## 代码准备 + +```python +from transformers import AutoTokenizer,AutoModelForCausalLM +import torch +# 使用模型下载到的本地路径以加载 +model_dir = '/root/autodl-tmp/THUCoAI/CharacterGLM-6B' +# 分词器的加载,本地加载,trust_remote_code=True设置允许从网络上下载模型权重和相关的代码 +tokenizer = AutoTokenizer.from_pretrained(model_dir, trust_remote_code=True) +# 模型加载,本地加载,使用AutoModelForCausalLM类 +model = AutoModelForCausalLM.from_pretrained(model_dir, trust_remote_code=True) +# 将模型移动到GPU上进行加速(如果有GPU的话) +device = torch.device("cuda" if torch.cuda.is_available() else "cpu") +model.to(device) +# 使用模型的评估模式来产生对话 +model.eval() +session_meta = {'user_info': '我是陆星辰,是一个男性,是一位知名导演,也是苏梦远的合作导演。我擅长拍摄音乐题材的电影。苏梦远对我的态度是尊敬的,并视我为良师益友。', 'bot_info': '苏梦远,本名苏远心,是一位当红的国内女歌手及演员。在参加选秀节目后,凭借独特的嗓音及出众的舞台魅力迅速成名,进入娱乐圈。她外表美丽动人,但真正的魅力在于她的才华和勤奋。苏梦远是音乐学院毕业的优秀生,善于创作,拥有多首热门原创歌曲。除了音乐方面的成就,她还热衷于慈善事业,积极参加公益活动,用实际行动传递正能量。在工作中,她对待工作非常敬业,拍戏时总是全身心投入角色,赢得了业内人士的赞誉和粉丝的喜爱。虽然在娱乐圈,但她始终保持低调、谦逊的态度,深得同行尊重。在表达时,苏梦远喜欢使用“我们”和“一起”,强调团队精神。', 'bot_name': '苏梦远', 'user_name': '陆星辰'} +# 第一轮对话 +response, history = model.chat(tokenizer, session_meta,"你好呀,小苏", history=[]) +print(response) +# 第二轮对话 +response, history = model.chat(tokenizer, session_meta,"最近对音乐有什么新的想法吗", history=history) +print(response) +# 第三轮对话 +response, history = model.chat(tokenizer,session_meta, "那我们商量一下下一部音乐电影的拍摄,好嘛?", history=history) +print(response) +``` + +## 部署 + +在终端输入以下命令运行trans.py,即实现CharacterGLM-6B的Transformers部署调用 + +```python +cd /root/autodl-tmp +python trans.py +``` + +观察命令行中loading checkpoint表示模型正在加载,等待模型加载完成产生对话,如下图所示 + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/f9d65275-fa89-4039-95c5-7cc0615753e2) + diff --git a/CharacterGLM/02-CharacterGLM-6B FastApi部署调用.md b/models/CharacterGLM/02-CharacterGLM-6B FastApi部署调用.md similarity index 97% rename from CharacterGLM/02-CharacterGLM-6B FastApi部署调用.md rename to models/CharacterGLM/02-CharacterGLM-6B FastApi部署调用.md index a0d6a07..ae20890 100644 --- a/CharacterGLM/02-CharacterGLM-6B FastApi部署调用.md +++ b/models/CharacterGLM/02-CharacterGLM-6B FastApi部署调用.md @@ -1,183 +1,183 @@ -# CharacterGLM-6B FastApi部署调用 - -## 环境准备 - -在autodl平台中租一个3090等24G显存的显卡机器,如下图所示镜像选择PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/2992ca12-7566-4916-94a6-1367df1a0d35) - - -接下来打开刚刚租用服务器的JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行demo。 - -pip换源和安装依赖包 - -```python -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install fastapi==0.104.1 -pip install uvicorn==0.24.0.post1 -pip install requests==2.25.1 -pip install modelscope==1.9.5 -pip install transformers==4.37.2 -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.24.1 - -``` - -## 模型下载 - -使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 python /root/autodl-tmp/download.py执行下载,模型大小为 12 GB,下载模型大概需要 10~15 分钟 - -```python -import torch -from modelscope import snapshot_download, AutoModel, AutoTokenizer -import os -model_dir = snapshot_download('THUCoAI/CharacterGLM-6B', cache_dir='/root/autodl-tmp', revision='master') -``` - -## 代码准备 - -在/root/autodl-tmp路径下新建api.py文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 - -```python -from fastapi import FastAPI, Request -from transformers import AutoTokenizer, AutoModelForCausalLM -import uvicorn -import json -import datetime -import torch - -# 设置设备参数 -DEVICE = "cuda" # 使用CUDA -DEVICE_ID = "0" # CUDA设备ID,如果未设置则为空 -CUDA_DEVICE = f"{DEVICE}:{DEVICE_ID}" if DEVICE_ID else DEVICE # 组合CUDA设备信息 - -# 清理GPU内存函数 -def torch_gc(): - if torch.cuda.is_available(): # 检查是否可用CUDA - with torch.cuda.device(CUDA_DEVICE): # 指定CUDA设备 - torch.cuda.empty_cache() # 清空CUDA缓存 - torch.cuda.ipc_collect() # 收集CUDA内存碎片 - -# 创建FastAPI应用 -app = FastAPI() - -# 处理POST请求的端点 -@app.post("/") -async def create_item(request: Request): - global model, tokenizer # 声明全局变量以便在函数内部使用模型和分词器 - json_post_raw = await request.json() # 获取POST请求的JSON数据 - json_post = json.dumps(json_post_raw) # 将JSON数据转换为字符串 - json_post_list = json.loads(json_post) # 将字符串转换为Python对象 - prompt = json_post_list.get('prompt') # 获取请求中的提示 - history = json_post_list.get('history') # 获取请求中的历史记录 - max_length = json_post_list.get('max_length') # 获取请求中的最大长度 - top_p = json_post_list.get('top_p') # 获取请求中的top_p参数 - temperature = json_post_list.get('temperature') # 获取请求中的温度参数 - session_meta = {'user_info': '我是陆星辰,是一个男性,是一位知名导演,也是苏梦远的合作导演。我擅长拍摄音乐题材的电影。苏梦远对我的态度是尊敬的,并视我为良师益友。', 'bot_info': '苏梦远,本名苏远心,是一位当红的国内女歌手及演员。在参加选秀节目后,凭借独特的嗓音及出众的舞台魅力迅速成名,进入娱乐圈。她外表美丽动人,但真正的魅力在于她的才华和勤奋。苏梦远是音乐学院毕业的优秀生,善于创作,拥有多首热门原创歌曲。除了音乐方面的成就,她还热衷于慈善事业,积极参加公益活动,用实际行动传递正能量。在工作中,她对待工作非常敬业,拍戏时总是全身心投入角色,赢得了业内人士的赞誉和粉丝的喜爱。虽然在娱乐圈,但她始终保持低调、谦逊的态度,深得同行尊重。在表达时,苏梦远喜欢使用“我们”和“一起”,强调团队精神。', 'bot_name': '苏梦远', 'user_name': '陆星辰'} - # 调用模型进行对话生成 - response, history = model.chat( - tokenizer, - session_meta, - prompt, - history=history, - max_length=max_length if max_length else 2048, # 如果未提供最大长度,默认使用2048 - top_p=top_p if top_p else 0.7, # 如果未提供top_p参数,默认使用0.7 - temperature=temperature if temperature else 0.95 # 如果未提供温度参数,默认使用0.95 - ) - now = datetime.datetime.now() # 获取当前时间 - time = now.strftime("%Y-%m-%d %H:%M:%S") # 格式化时间为字符串 - # 构建响应JSON - answer = { - "response": response, - "history": history, - "status": 200, - "time": time - } - # 构建日志信息 - log = "[" + time + "] " + '", prompt:"' + prompt + '", response:"' + repr(response) + '"' - print(log) # 打印日志 - torch_gc() # 执行GPU内存清理 - return answer # 返回响应 - -# 主函数入口 -if __name__ == '__main__': - # 加载预训练的分词器和模型 - tokenizer = AutoTokenizer.from_pretrained("/root/autodl-tmp/THUCoAI/CharacterGLM-6B", trust_remote_code=True) - model = AutoModelForCausalLM.from_pretrained("/root/autodl-tmp/THUCoAI/CharacterGLM-6B", trust_remote_code=True).to(torch.bfloat16).cuda() - model.eval() # 设置模型为评估模式 - # 启动FastAPI应用 - # 用6006端口可以将autodl的端口映射到本地,从而在本地使用api - uvicorn.run(app, host='0.0.0.0', port=6006, workers=1) # 在指定端口和主机上启动应用 -``` - -## Api部署调用 - -在终端输入以下命令启动api服务 - -```python - -cd /root/autodl-tmp -python api.py - -``` - -默认部署在 6006 端口,通过 POST 方法进行调用,可以使用curl调用,如下所示: - -```python - -curl -X POST "http://127.0.0.1:6006" \ - -H 'Content-Type: application/json' \ - -d '{"prompt": "你好", "history": []}' - -``` - -调用示例结果如下图所示 - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/c5568fe4-ae2a-4679-b795-f3fe17458310) - - -也可以使用python中的requests库进行调用,新建api-requests.py文件,写入如下代码: - -```python - -import requests -import json - -def get_completion(prompt): - headers = {'Content-Type': 'application/json'} - data = {"prompt": prompt, "history": []} - response = requests.post(url='http://127.0.0.1:6006', headers=headers, data=json.dumps(data)) - return response.json()['response'] - -if __name__ == '__main__': - print(get_completion('你是谁呀?')) - -``` - -新开一个终端,输入如下指令 - -```python -cd /root/autodl-tmp -python api-requests.py - -``` - -得到的返回值及结果展示如下 - -```python -{ -'response': '嗨,你好,我叫苏梦远。(微笑着向对方走去)', -'history': [['你是谁呀?', '嗨,你好,我叫苏梦远。(微笑着向对方走去)']], -'status': 200, -'time': '2024-03-05 22:44:35' -} -``` - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/f14d739e-addf-4b1b-bcf3-da4d714130fe) +# CharacterGLM-6B FastApi部署调用 + +## 环境准备 + +在autodl平台中租一个3090等24G显存的显卡机器,如下图所示镜像选择PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/2992ca12-7566-4916-94a6-1367df1a0d35) + + +接下来打开刚刚租用服务器的JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行demo。 + +pip换源和安装依赖包 + +```python +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install fastapi==0.104.1 +pip install uvicorn==0.24.0.post1 +pip install requests==2.25.1 +pip install modelscope==1.9.5 +pip install transformers==4.37.2 +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.24.1 + +``` + +## 模型下载 + +使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 python /root/autodl-tmp/download.py执行下载,模型大小为 12 GB,下载模型大概需要 10~15 分钟 + +```python +import torch +from modelscope import snapshot_download, AutoModel, AutoTokenizer +import os +model_dir = snapshot_download('THUCoAI/CharacterGLM-6B', cache_dir='/root/autodl-tmp', revision='master') +``` + +## 代码准备 + +在/root/autodl-tmp路径下新建api.py文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 + +```python +from fastapi import FastAPI, Request +from transformers import AutoTokenizer, AutoModelForCausalLM +import uvicorn +import json +import datetime +import torch + +# 设置设备参数 +DEVICE = "cuda" # 使用CUDA +DEVICE_ID = "0" # CUDA设备ID,如果未设置则为空 +CUDA_DEVICE = f"{DEVICE}:{DEVICE_ID}" if DEVICE_ID else DEVICE # 组合CUDA设备信息 + +# 清理GPU内存函数 +def torch_gc(): + if torch.cuda.is_available(): # 检查是否可用CUDA + with torch.cuda.device(CUDA_DEVICE): # 指定CUDA设备 + torch.cuda.empty_cache() # 清空CUDA缓存 + torch.cuda.ipc_collect() # 收集CUDA内存碎片 + +# 创建FastAPI应用 +app = FastAPI() + +# 处理POST请求的端点 +@app.post("/") +async def create_item(request: Request): + global model, tokenizer # 声明全局变量以便在函数内部使用模型和分词器 + json_post_raw = await request.json() # 获取POST请求的JSON数据 + json_post = json.dumps(json_post_raw) # 将JSON数据转换为字符串 + json_post_list = json.loads(json_post) # 将字符串转换为Python对象 + prompt = json_post_list.get('prompt') # 获取请求中的提示 + history = json_post_list.get('history') # 获取请求中的历史记录 + max_length = json_post_list.get('max_length') # 获取请求中的最大长度 + top_p = json_post_list.get('top_p') # 获取请求中的top_p参数 + temperature = json_post_list.get('temperature') # 获取请求中的温度参数 + session_meta = {'user_info': '我是陆星辰,是一个男性,是一位知名导演,也是苏梦远的合作导演。我擅长拍摄音乐题材的电影。苏梦远对我的态度是尊敬的,并视我为良师益友。', 'bot_info': '苏梦远,本名苏远心,是一位当红的国内女歌手及演员。在参加选秀节目后,凭借独特的嗓音及出众的舞台魅力迅速成名,进入娱乐圈。她外表美丽动人,但真正的魅力在于她的才华和勤奋。苏梦远是音乐学院毕业的优秀生,善于创作,拥有多首热门原创歌曲。除了音乐方面的成就,她还热衷于慈善事业,积极参加公益活动,用实际行动传递正能量。在工作中,她对待工作非常敬业,拍戏时总是全身心投入角色,赢得了业内人士的赞誉和粉丝的喜爱。虽然在娱乐圈,但她始终保持低调、谦逊的态度,深得同行尊重。在表达时,苏梦远喜欢使用“我们”和“一起”,强调团队精神。', 'bot_name': '苏梦远', 'user_name': '陆星辰'} + # 调用模型进行对话生成 + response, history = model.chat( + tokenizer, + session_meta, + prompt, + history=history, + max_length=max_length if max_length else 2048, # 如果未提供最大长度,默认使用2048 + top_p=top_p if top_p else 0.7, # 如果未提供top_p参数,默认使用0.7 + temperature=temperature if temperature else 0.95 # 如果未提供温度参数,默认使用0.95 + ) + now = datetime.datetime.now() # 获取当前时间 + time = now.strftime("%Y-%m-%d %H:%M:%S") # 格式化时间为字符串 + # 构建响应JSON + answer = { + "response": response, + "history": history, + "status": 200, + "time": time + } + # 构建日志信息 + log = "[" + time + "] " + '", prompt:"' + prompt + '", response:"' + repr(response) + '"' + print(log) # 打印日志 + torch_gc() # 执行GPU内存清理 + return answer # 返回响应 + +# 主函数入口 +if __name__ == '__main__': + # 加载预训练的分词器和模型 + tokenizer = AutoTokenizer.from_pretrained("/root/autodl-tmp/THUCoAI/CharacterGLM-6B", trust_remote_code=True) + model = AutoModelForCausalLM.from_pretrained("/root/autodl-tmp/THUCoAI/CharacterGLM-6B", trust_remote_code=True).to(torch.bfloat16).cuda() + model.eval() # 设置模型为评估模式 + # 启动FastAPI应用 + # 用6006端口可以将autodl的端口映射到本地,从而在本地使用api + uvicorn.run(app, host='0.0.0.0', port=6006, workers=1) # 在指定端口和主机上启动应用 +``` + +## Api部署调用 + +在终端输入以下命令启动api服务 + +```python + +cd /root/autodl-tmp +python api.py + +``` + +默认部署在 6006 端口,通过 POST 方法进行调用,可以使用curl调用,如下所示: + +```python + +curl -X POST "http://127.0.0.1:6006" \ + -H 'Content-Type: application/json' \ + -d '{"prompt": "你好", "history": []}' + +``` + +调用示例结果如下图所示 + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/c5568fe4-ae2a-4679-b795-f3fe17458310) + + +也可以使用python中的requests库进行调用,新建api-requests.py文件,写入如下代码: + +```python + +import requests +import json + +def get_completion(prompt): + headers = {'Content-Type': 'application/json'} + data = {"prompt": prompt, "history": []} + response = requests.post(url='http://127.0.0.1:6006', headers=headers, data=json.dumps(data)) + return response.json()['response'] + +if __name__ == '__main__': + print(get_completion('你是谁呀?')) + +``` + +新开一个终端,输入如下指令 + +```python +cd /root/autodl-tmp +python api-requests.py + +``` + +得到的返回值及结果展示如下 + +```python +{ +'response': '嗨,你好,我叫苏梦远。(微笑着向对方走去)', +'history': [['你是谁呀?', '嗨,你好,我叫苏梦远。(微笑着向对方走去)']], +'status': 200, +'time': '2024-03-05 22:44:35' +} +``` + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/f14d739e-addf-4b1b-bcf3-da4d714130fe) diff --git a/CharacterGLM/03-CharacterGLM-6B-chat.md b/models/CharacterGLM/03-CharacterGLM-6B-chat.md similarity index 97% rename from CharacterGLM/03-CharacterGLM-6B-chat.md rename to models/CharacterGLM/03-CharacterGLM-6B-chat.md index fd72b9e..45be370 100644 --- a/CharacterGLM/03-CharacterGLM-6B-chat.md +++ b/models/CharacterGLM/03-CharacterGLM-6B-chat.md @@ -1,96 +1,96 @@ -# CharacterGLM-6B-chat - -## 环境准备 - -在autodl平台中租一个3090等24G显存的显卡机器,如下图所示镜像选择PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/0dddbee9-df80-4033-9568-185ea585f261) - - -接下来打开刚刚租用服务器的JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行demo。 - -pip换源和安装依赖包 - -```python -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install modelscope -pip install transformers -``` - -## 模型下载 - -使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 python /root/autodl-tmp/download.py执行下载,模型大小为 12 GB,下载模型大概需要 10~15 分钟 - -```python -import torch -from modelscope import snapshot_download, AutoModel, AutoTokenizer -import os -model_dir = snapshot_download('THUCoAI/CharacterGLM-6B', cache_dir='/root/autodl-tmp', revision='master') -``` - -## 代码准备 - -首先clone代码,打开autodl平台自带的学术镜像加速。学术镜像加速详细使用请看: -https://www.autodl.com/docs/network_turbo/ - -```python -source /etc/network_turbo -``` - -然后切换路径, clone代码. - -```python -cd /root/autodl-tmp -git clone https://github.com/thu-coai/CharacterGLM-6B -``` - -## demo运行 - -修改代码路径,将 /root/autodl-tmp/CharacterGLM-6B/basic_demo/web_demo_streamlit.py中第20行的模型更换为本地的/root/autodl-tmp/THUCoAI/CharacterGLM-6B - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/1edc97a2-3d6e-43e3-b176-644b756b615f) - - -修改requirements.txt文件,将其中的torch删掉,环境中已经有了torch,不需要再安装。然后执行下面的命令: - -```python -cd /root/autodl-tmp/CharacterGLM-6B -pip install -r requirements.txt -``` - -在终端运行以下命令即可启动推理服务,尽量cd到basic_demo文件夹下,防止找不到character.json文件 - -```python -cd /root/autodl-tmp/CharacterGLM-6B/basic_demo -streamlit run ./web_demo2.py --server.address 127.0.0.1 --server.port 6006 -``` - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/2fff8bd4-6d4b-449f-81ee-dc9e42b8ceb8) - - -在将 autodl 的端口映射到本地的 http://localhost:6006 后,即可看到demo界面。具体映射步骤参考文档General-Setting文件夹下/02-AutoDL开放端口.md文档。 - -在浏览器打开 http://localhost:6006 界面,模型加载,即可使用,如下图所示。 - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/ac7a9887-4628-4539-9297-caccfb523530) - - -## 命令行运行 - -修改代码路径,将 /root/autodl-tmp/CharacterGLM-6B/basic_demo/cli_demo.py中的模型路径更换为本地的/root/autodl-tmp/THUCoAI/CharacterGLM-6B - -在终端运行以下命令即可启动推理服务 - -```python -cd /root/autodl-tmp/CharacterGLM-6B/basic_demo -python ./cli_demo.py -``` - -![image](https://github.com/suncaleb1/self-llm/assets/155936975/1eb29dd5-8bae-458f-908f-f7388ae248c0) - +# CharacterGLM-6B-chat + +## 环境准备 + +在autodl平台中租一个3090等24G显存的显卡机器,如下图所示镜像选择PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/0dddbee9-df80-4033-9568-185ea585f261) + + +接下来打开刚刚租用服务器的JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行demo。 + +pip换源和安装依赖包 + +```python +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install modelscope +pip install transformers +``` + +## 模型下载 + +使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 python /root/autodl-tmp/download.py执行下载,模型大小为 12 GB,下载模型大概需要 10~15 分钟 + +```python +import torch +from modelscope import snapshot_download, AutoModel, AutoTokenizer +import os +model_dir = snapshot_download('THUCoAI/CharacterGLM-6B', cache_dir='/root/autodl-tmp', revision='master') +``` + +## 代码准备 + +首先clone代码,打开autodl平台自带的学术镜像加速。学术镜像加速详细使用请看: +https://www.autodl.com/docs/network_turbo/ + +```python +source /etc/network_turbo +``` + +然后切换路径, clone代码. + +```python +cd /root/autodl-tmp +git clone https://github.com/thu-coai/CharacterGLM-6B +``` + +## demo运行 + +修改代码路径,将 /root/autodl-tmp/CharacterGLM-6B/basic_demo/web_demo_streamlit.py中第20行的模型更换为本地的/root/autodl-tmp/THUCoAI/CharacterGLM-6B + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/1edc97a2-3d6e-43e3-b176-644b756b615f) + + +修改requirements.txt文件,将其中的torch删掉,环境中已经有了torch,不需要再安装。然后执行下面的命令: + +```python +cd /root/autodl-tmp/CharacterGLM-6B +pip install -r requirements.txt +``` + +在终端运行以下命令即可启动推理服务,尽量cd到basic_demo文件夹下,防止找不到character.json文件 + +```python +cd /root/autodl-tmp/CharacterGLM-6B/basic_demo +streamlit run ./web_demo2.py --server.address 127.0.0.1 --server.port 6006 +``` + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/2fff8bd4-6d4b-449f-81ee-dc9e42b8ceb8) + + +在将 autodl 的端口映射到本地的 http://localhost:6006 后,即可看到demo界面。具体映射步骤参考文档General-Setting文件夹下/02-AutoDL开放端口.md文档。 + +在浏览器打开 http://localhost:6006 界面,模型加载,即可使用,如下图所示。 + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/ac7a9887-4628-4539-9297-caccfb523530) + + +## 命令行运行 + +修改代码路径,将 /root/autodl-tmp/CharacterGLM-6B/basic_demo/cli_demo.py中的模型路径更换为本地的/root/autodl-tmp/THUCoAI/CharacterGLM-6B + +在终端运行以下命令即可启动推理服务 + +```python +cd /root/autodl-tmp/CharacterGLM-6B/basic_demo +python ./cli_demo.py +``` + +![image](https://github.com/suncaleb1/self-llm/assets/155936975/1eb29dd5-8bae-458f-908f-f7388ae248c0) + diff --git a/CharacterGLM/04-CharacterGLM-6B Lora微调.md b/models/CharacterGLM/04-CharacterGLM-6B Lora微调.md similarity index 97% rename from CharacterGLM/04-CharacterGLM-6B Lora微调.md rename to models/CharacterGLM/04-CharacterGLM-6B Lora微调.md index 281f6af..d5202bf 100644 --- a/CharacterGLM/04-CharacterGLM-6B Lora微调.md +++ b/models/CharacterGLM/04-CharacterGLM-6B Lora微调.md @@ -1,195 +1,195 @@ -# 04-CharacterGLM-6B-Chat Lora微调 - -## 概述 - -本文简要介绍如何基于transformers、peft等框架,对CharacterGLM-6B-chat模型进行Lora微调。Lora原理可参考博客:[知乎|深入浅出Lora](https://zhuanlan.zhihu.com/p/650197598) -本文代码未使用分布式框架,微调 ChatGLM3-6B-Chat 模型至少需要 21G 及以上的显存,且需要修改脚本文件中的模型路径和数据集路径。 - -## 环境配置 - -在完成基本环境配置和本地模型部署的情况下,还需要安装一些第三方库,可以使用如下命令: - -```python -pip install transformers==4.37.2 -pip install peft==0.4.0.dev0 -pip install datasets==2.10.1 -pip install accelerate==0.21.0 - -``` - -在本节内容中,将微调数据集放置在根目录[/dataset](https://github.com/datawhalechina/self-llm/blob/master/dataset/huanhuan.json)。 - -## 指令集构建 - -LLM微调一般指指令微调过程。所谓指令微调,是说我们使用的微调数据形如: - -```python -{ - "instruction":"回答用户以下问题,直接给出结果。" - "input":"中国第一个诺贝尔奖得主是谁?" - "output":"莫言" -} -``` - -其中instruction是用户指令,告知模型需要完成的任务;input是用户输入,是完成用户指令所必需的输入内容;output是模型应该给出的输出。 - -即我们的核心训练目标是让模型具有理解并遵循用户指令的能力。因此,在指令集构建时,我们应针对我们的目标任务,针对性构建任务指令集。在本文我们使用由笔者合作开源的[Chat-甄嬛项目](https://github.com/KMnO4-zx/huanhuan-chat)作为示例,我们的目标是构建一个能够模拟甄嬛对话风格的个性化LLM,因此我们构建的指令形如: - -```python -{ - "instruction": "", - "input":"你是谁?", - "output":"家父是大理寺少卿甄远道。" -} -``` - -我们构造的全部指令数据集在根目录下。 - -## QA和Instruction的区别和联系 - -QA是指一问一答的形式,通常是用户提问,模型给出回答。而instruction则源自于Prompt Engineering,将问题拆分成两个部分:Instruction用于描述任务,Input用于描述待处理的对象。 - -问答(QA)格式的训练数据通常用于训练模型执行具体任务。例如,对于问题“请解释INFJ和ENTP两种MBTI性格之间的区别” - -*问答(QA)格式: - -```python -指令(instruction): -输入(input):INFJ和ENTP这两种MBTI性格之间的区别是什么? -``` - -*指令(Instruction)格式: - -```python -指令(Instruction):请解释下面两种MBTI性格的区别 -输入(input):INFJ和ENTP -``` - -## 数据格式化 - -Lora训练的数据是需要经过格式化、编码之后再输入给模型进行训练的,我们一般需要将输入文本编码为input_ids,将输出文本编码为labels,编码之后的结果都是多维向量。我们首先定义一个与处理函数,这个函数用于对每一个样本,编码其输入,输出文本并返回一个编码后的字典: - -```python -def process_func(example): - MAX_LENGTH = 512 - input_ids, labels = [], [] - prompt = tokenizer.encode("用户:\n"+"现在你要扮演皇帝身边的女人--甄嬛。", add_special_tokens=False) - instruction_ = tokenizer.encode("\n".join([example["instruction"], example["input"]]).strip(), add_special_tokens=False,max_length=512) - instruction = tokenizer.encode(prompt + instruction_) - response = tokenizer.encode("CharacterGLM-6B:\n:" + example["output"], add_special_tokens=False) - input_ids = instruction + response + [tokenizer.eos_token_id] - labels = [tokenizer.pad_token_id] * len(instruction) + response + [tokenizer.eos_token_id] - pad_len = MAX_LENGTH - len(input_ids) - # print() - input_ids += [tokenizer.pad_token_id] * pad_len - labels += [tokenizer.pad_token_id] * pad_len - labels = [(l if l != tokenizer.pad_token_id else -100) for l in labels] - - return { - "input_ids": input_ids, - "labels": labels - } -``` - -经过格式化的数据,也就是送入模型的每一条数据,都是一个字典,包含了input_ids、labels两个键值对,其中input_ids是输入文本的编码,labels是输出文本的编码。 - -## 加载tokenizer和半精度模型 - -模型以版精度形式加载,如果显卡比较新,可以用torch.bfloat形式加载,对于自定义的模型一定要指定trust_remote_code参数为True - -```python -tokenizer=AutoTokenizer.from_pretrained('/root/autodl-tmp/THUCoAI/CharacterGLM-6B',use_fast=False,trust_remote_code=True) - -model=AutoModelForCausalLM.from_pretrained('/root/autodl-tmp/THUCoAI/CharacterGLM-6B',trust_remote_code=True,torch_dtype=torch.half,device_map="auto") -``` - -## 定义LoraConfig - -LoraConfig这个类中可以设置很多参数,部分参数展示如下: -task_type:模型类型 -target——modules:需要训练的模型层的名字,主要就是attention部分的层,不同的模型对应的层的名字不同,可以传入数组,也可以字符串,也可以正则表达式。 -r:lora的秩 -lora_alpha:Lora alpha -modules_to_save:指定的是除了拆成lora的模块,其它的模块可以完整的指定训练 - -Lora的所方式lora_alpha/r,在这个LoraConfig中缩放就是4倍。这个缩放的本质并没有改变Lora的参数量大小,本质在于将里面的参数数值做广播乘法,进行线性的缩放。 - -```python -config=LoraConfig( - task_type=TaskType.CAUSAL_LM, - target_modules=["query_key_value"], - inference_mode=False, - r=8, - lora_alpha=32, - lora_dropout=0.1 -) -``` - -## 自定义TraininArguments参数 - -TrainingArguments这个类的源码也介绍了每个参数的具体作用,常用的参数如下: -output_dir:模型的输出路径 -per_device_train_batch_size:batch_size -gradient_accumulation_steps:梯度累加,如果显存比较小,可以把batch_size设置小一点,梯度累积增大一点 -logging_steps:多少步,输出一次log -num_train_epochs:顾名思义epoch -gradient_chechpointing:梯度检查,这个一旦开启,模型就必须执行 -model.enable_input_require_grads() - -```python -data_collator=DataCollatorForSeq2Seq( - tokenizer, - model=model, - label_pad_token_id=-100, - pad_to_multiple_of=None, - padding=False -) -args=TrainingArguments( - output_dir="./output/CharacterGLM", - per_device_train_batch_size=4, - gradient_accumulation_steps=2, - logging_steps=10, - num_train_epochs=3, - gradient_checkpointing=True, - save_steps=100, - learning_rate=1e-4, -) -``` - -## 使用Trainer训练 - -把model放进去,把上面设置的参数放进去,数据集放进去,开始训练 - -```python -trainer=Trainer( - model=model, - args=args, - train_dataset=tokenized_id, - data_collator=data_collator, -) -trainer.train() -``` - -## 模型推理 - -```python -model = model.cuda() -ipt = tokenizer("用户:{}\n{}".format("现在你要扮演皇帝身边的女人--甄嬛。你是谁?", "").strip() + "characterGLM-6B:\n", return_tensors="pt").to(model.device) -tokenizer.decode(model.generate(**ipt, max_length=128, do_sample=True)[0], skip_special_tokens=True) -``` - -## 从新加载 - -通过PEFT所微调的模型,都可以使用下面的方法进行重新加载,并推理: - -加载源model与tokenizer; -使用PeftModel合并源model与PEFT微调后的参数 - -```python -from peft import Peftmodel -model=AutoModelForCausalLM.from_pretrained("/root/autodl-tmp/THUCoAI/CharacterGLM-6B",trust_remote_code=True,low_cpu_mem_usage=True) -tokenizer=AutoTokenizer.from_pretrained("root/autodl-tmp/THUCoAI/CharacterGLM-6B",use_fast=False,trust_remote_code=True) -p_model=PeftModel.from_pretrained(model,model_id="./output/CharatcerGLM/checkpoint-1000/") -ipt = tokenizer("用户:{}\n{}".format("现在你要扮演皇帝身边的女人--甄嬛。你是谁?", "").strip() + "characterGLM-6B:\n", return_tensors="pt").to(model.device) -tokenizer.decode(p_model.generate(**ipt,max_length=128,do_sample=True)[0],skip_special_tokens=True) -``` +# 04-CharacterGLM-6B-Chat Lora微调 + +## 概述 + +本文简要介绍如何基于transformers、peft等框架,对CharacterGLM-6B-chat模型进行Lora微调。Lora原理可参考博客:[知乎|深入浅出Lora](https://zhuanlan.zhihu.com/p/650197598) +本文代码未使用分布式框架,微调 ChatGLM3-6B-Chat 模型至少需要 21G 及以上的显存,且需要修改脚本文件中的模型路径和数据集路径。 + +## 环境配置 + +在完成基本环境配置和本地模型部署的情况下,还需要安装一些第三方库,可以使用如下命令: + +```python +pip install transformers==4.37.2 +pip install peft==0.4.0.dev0 +pip install datasets==2.10.1 +pip install accelerate==0.21.0 + +``` + +在本节内容中,将微调数据集放置在根目录[/dataset](https://github.com/datawhalechina/self-llm/blob/master/dataset/huanhuan.json)。 + +## 指令集构建 + +LLM微调一般指指令微调过程。所谓指令微调,是说我们使用的微调数据形如: + +```python +{ + "instruction":"回答用户以下问题,直接给出结果。" + "input":"中国第一个诺贝尔奖得主是谁?" + "output":"莫言" +} +``` + +其中instruction是用户指令,告知模型需要完成的任务;input是用户输入,是完成用户指令所必需的输入内容;output是模型应该给出的输出。 + +即我们的核心训练目标是让模型具有理解并遵循用户指令的能力。因此,在指令集构建时,我们应针对我们的目标任务,针对性构建任务指令集。在本文我们使用由笔者合作开源的[Chat-甄嬛项目](https://github.com/KMnO4-zx/huanhuan-chat)作为示例,我们的目标是构建一个能够模拟甄嬛对话风格的个性化LLM,因此我们构建的指令形如: + +```python +{ + "instruction": "", + "input":"你是谁?", + "output":"家父是大理寺少卿甄远道。" +} +``` + +我们构造的全部指令数据集在根目录下。 + +## QA和Instruction的区别和联系 + +QA是指一问一答的形式,通常是用户提问,模型给出回答。而instruction则源自于Prompt Engineering,将问题拆分成两个部分:Instruction用于描述任务,Input用于描述待处理的对象。 + +问答(QA)格式的训练数据通常用于训练模型执行具体任务。例如,对于问题“请解释INFJ和ENTP两种MBTI性格之间的区别” + +*问答(QA)格式: + +```python +指令(instruction): +输入(input):INFJ和ENTP这两种MBTI性格之间的区别是什么? +``` + +*指令(Instruction)格式: + +```python +指令(Instruction):请解释下面两种MBTI性格的区别 +输入(input):INFJ和ENTP +``` + +## 数据格式化 + +Lora训练的数据是需要经过格式化、编码之后再输入给模型进行训练的,我们一般需要将输入文本编码为input_ids,将输出文本编码为labels,编码之后的结果都是多维向量。我们首先定义一个与处理函数,这个函数用于对每一个样本,编码其输入,输出文本并返回一个编码后的字典: + +```python +def process_func(example): + MAX_LENGTH = 512 + input_ids, labels = [], [] + prompt = tokenizer.encode("用户:\n"+"现在你要扮演皇帝身边的女人--甄嬛。", add_special_tokens=False) + instruction_ = tokenizer.encode("\n".join([example["instruction"], example["input"]]).strip(), add_special_tokens=False,max_length=512) + instruction = tokenizer.encode(prompt + instruction_) + response = tokenizer.encode("CharacterGLM-6B:\n:" + example["output"], add_special_tokens=False) + input_ids = instruction + response + [tokenizer.eos_token_id] + labels = [tokenizer.pad_token_id] * len(instruction) + response + [tokenizer.eos_token_id] + pad_len = MAX_LENGTH - len(input_ids) + # print() + input_ids += [tokenizer.pad_token_id] * pad_len + labels += [tokenizer.pad_token_id] * pad_len + labels = [(l if l != tokenizer.pad_token_id else -100) for l in labels] + + return { + "input_ids": input_ids, + "labels": labels + } +``` + +经过格式化的数据,也就是送入模型的每一条数据,都是一个字典,包含了input_ids、labels两个键值对,其中input_ids是输入文本的编码,labels是输出文本的编码。 + +## 加载tokenizer和半精度模型 + +模型以版精度形式加载,如果显卡比较新,可以用torch.bfloat形式加载,对于自定义的模型一定要指定trust_remote_code参数为True + +```python +tokenizer=AutoTokenizer.from_pretrained('/root/autodl-tmp/THUCoAI/CharacterGLM-6B',use_fast=False,trust_remote_code=True) + +model=AutoModelForCausalLM.from_pretrained('/root/autodl-tmp/THUCoAI/CharacterGLM-6B',trust_remote_code=True,torch_dtype=torch.half,device_map="auto") +``` + +## 定义LoraConfig + +LoraConfig这个类中可以设置很多参数,部分参数展示如下: +task_type:模型类型 +target——modules:需要训练的模型层的名字,主要就是attention部分的层,不同的模型对应的层的名字不同,可以传入数组,也可以字符串,也可以正则表达式。 +r:lora的秩 +lora_alpha:Lora alpha +modules_to_save:指定的是除了拆成lora的模块,其它的模块可以完整的指定训练 + +Lora的所方式lora_alpha/r,在这个LoraConfig中缩放就是4倍。这个缩放的本质并没有改变Lora的参数量大小,本质在于将里面的参数数值做广播乘法,进行线性的缩放。 + +```python +config=LoraConfig( + task_type=TaskType.CAUSAL_LM, + target_modules=["query_key_value"], + inference_mode=False, + r=8, + lora_alpha=32, + lora_dropout=0.1 +) +``` + +## 自定义TraininArguments参数 + +TrainingArguments这个类的源码也介绍了每个参数的具体作用,常用的参数如下: +output_dir:模型的输出路径 +per_device_train_batch_size:batch_size +gradient_accumulation_steps:梯度累加,如果显存比较小,可以把batch_size设置小一点,梯度累积增大一点 +logging_steps:多少步,输出一次log +num_train_epochs:顾名思义epoch +gradient_chechpointing:梯度检查,这个一旦开启,模型就必须执行 +model.enable_input_require_grads() + +```python +data_collator=DataCollatorForSeq2Seq( + tokenizer, + model=model, + label_pad_token_id=-100, + pad_to_multiple_of=None, + padding=False +) +args=TrainingArguments( + output_dir="./output/CharacterGLM", + per_device_train_batch_size=4, + gradient_accumulation_steps=2, + logging_steps=10, + num_train_epochs=3, + gradient_checkpointing=True, + save_steps=100, + learning_rate=1e-4, +) +``` + +## 使用Trainer训练 + +把model放进去,把上面设置的参数放进去,数据集放进去,开始训练 + +```python +trainer=Trainer( + model=model, + args=args, + train_dataset=tokenized_id, + data_collator=data_collator, +) +trainer.train() +``` + +## 模型推理 + +```python +model = model.cuda() +ipt = tokenizer("用户:{}\n{}".format("现在你要扮演皇帝身边的女人--甄嬛。你是谁?", "").strip() + "characterGLM-6B:\n", return_tensors="pt").to(model.device) +tokenizer.decode(model.generate(**ipt, max_length=128, do_sample=True)[0], skip_special_tokens=True) +``` + +## 从新加载 + +通过PEFT所微调的模型,都可以使用下面的方法进行重新加载,并推理: + +加载源model与tokenizer; +使用PeftModel合并源model与PEFT微调后的参数 + +```python +from peft import Peftmodel +model=AutoModelForCausalLM.from_pretrained("/root/autodl-tmp/THUCoAI/CharacterGLM-6B",trust_remote_code=True,low_cpu_mem_usage=True) +tokenizer=AutoTokenizer.from_pretrained("root/autodl-tmp/THUCoAI/CharacterGLM-6B",use_fast=False,trust_remote_code=True) +p_model=PeftModel.from_pretrained(model,model_id="./output/CharatcerGLM/checkpoint-1000/") +ipt = tokenizer("用户:{}\n{}".format("现在你要扮演皇帝身边的女人--甄嬛。你是谁?", "").strip() + "characterGLM-6B:\n", return_tensors="pt").to(model.device) +tokenizer.decode(p_model.generate(**ipt,max_length=128,do_sample=True)[0],skip_special_tokens=True) +``` diff --git a/CharacterGLM/04-CharacterGLM-6B-Lora微调.ipynb b/models/CharacterGLM/04-CharacterGLM-6B-Lora微调.ipynb similarity index 100% rename from CharacterGLM/04-CharacterGLM-6B-Lora微调.ipynb rename to models/CharacterGLM/04-CharacterGLM-6B-Lora微调.ipynb diff --git a/CharacterGLM/04-CharacterGLM-6B-Lora微调.py b/models/CharacterGLM/04-CharacterGLM-6B-Lora微调.py similarity index 100% rename from CharacterGLM/04-CharacterGLM-6B-Lora微调.py rename to models/CharacterGLM/04-CharacterGLM-6B-Lora微调.py diff --git a/CharacterGLM/image/03-webdemo_show.png b/models/CharacterGLM/image/03-webdemo_show.png similarity index 100% rename from CharacterGLM/image/03-webdemo_show.png rename to models/CharacterGLM/image/03-webdemo_show.png diff --git a/CharacterGLM/image/03-修改路径.png b/models/CharacterGLM/image/03-修改路径.png similarity index 100% rename from CharacterGLM/image/03-修改路径.png rename to models/CharacterGLM/image/03-修改路径.png diff --git a/CharacterGLM/image/03-运行clidemo.png b/models/CharacterGLM/image/03-运行clidemo.png similarity index 100% rename from CharacterGLM/image/03-运行clidemo.png rename to models/CharacterGLM/image/03-运行clidemo.png diff --git a/CharacterGLM/image/03-运行webdemo.png b/models/CharacterGLM/image/03-运行webdemo.png similarity index 100% rename from CharacterGLM/image/03-运行webdemo.png rename to models/CharacterGLM/image/03-运行webdemo.png diff --git a/CharacterGLM/image/image-1.png b/models/CharacterGLM/image/image-1.png similarity index 100% rename from CharacterGLM/image/image-1.png rename to models/CharacterGLM/image/image-1.png diff --git a/CharacterGLM/image/image-2.png b/models/CharacterGLM/image/image-2.png similarity index 100% rename from CharacterGLM/image/image-2.png rename to models/CharacterGLM/image/image-2.png diff --git a/CharacterGLM/image/image-3.png b/models/CharacterGLM/image/image-3.png similarity index 100% rename from CharacterGLM/image/image-3.png rename to models/CharacterGLM/image/image-3.png diff --git a/CharacterGLM/image/image-4.png b/models/CharacterGLM/image/image-4.png similarity index 100% rename from CharacterGLM/image/image-4.png rename to models/CharacterGLM/image/image-4.png diff --git a/CharacterGLM/image/readme.md b/models/CharacterGLM/image/readme.md similarity index 100% rename from CharacterGLM/image/readme.md rename to models/CharacterGLM/image/readme.md diff --git a/CharacterGLM/readme.md b/models/CharacterGLM/readme.md similarity index 100% rename from CharacterGLM/readme.md rename to models/CharacterGLM/readme.md diff --git a/ChatGLM/01-ChatGLM3-6B Transformer部署调用.md b/models/ChatGLM/01-ChatGLM3-6B Transformer部署调用.md similarity index 100% rename from ChatGLM/01-ChatGLM3-6B Transformer部署调用.md rename to models/ChatGLM/01-ChatGLM3-6B Transformer部署调用.md diff --git a/ChatGLM/02-ChatGLM3-6B FastApi部署调用.md b/models/ChatGLM/02-ChatGLM3-6B FastApi部署调用.md similarity index 100% rename from ChatGLM/02-ChatGLM3-6B FastApi部署调用.md rename to models/ChatGLM/02-ChatGLM3-6B FastApi部署调用.md diff --git a/ChatGLM/03-ChatGLM3-6B-chat.md b/models/ChatGLM/03-ChatGLM3-6B-chat.md similarity index 100% rename from ChatGLM/03-ChatGLM3-6B-chat.md rename to models/ChatGLM/03-ChatGLM3-6B-chat.md diff --git a/ChatGLM/04-ChatGLM3-6B-Code-Interpreter.md b/models/ChatGLM/04-ChatGLM3-6B-Code-Interpreter.md similarity index 100% rename from ChatGLM/04-ChatGLM3-6B-Code-Interpreter.md rename to models/ChatGLM/04-ChatGLM3-6B-Code-Interpreter.md diff --git a/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手.md b/models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手.md similarity index 100% rename from ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手.md rename to models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手.md diff --git a/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/LLM.py b/models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/LLM.py similarity index 100% rename from ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/LLM.py rename to models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/LLM.py diff --git a/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/create_db.py b/models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/create_db.py similarity index 100% rename from ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/create_db.py rename to models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/create_db.py diff --git a/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/run_gradio.py b/models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/run_gradio.py similarity index 100% rename from ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/run_gradio.py rename to models/ChatGLM/05-ChatGLM3-6B接入LangChain搭建知识库助手/run_gradio.py diff --git a/ChatGLM/06-ChatGLM3-6B-Lora微调.ipynb b/models/ChatGLM/06-ChatGLM3-6B-Lora微调.ipynb similarity index 100% rename from ChatGLM/06-ChatGLM3-6B-Lora微调.ipynb rename to models/ChatGLM/06-ChatGLM3-6B-Lora微调.ipynb diff --git a/ChatGLM/06-ChatGLM3-6B-Lora微调.md b/models/ChatGLM/06-ChatGLM3-6B-Lora微调.md similarity index 100% rename from ChatGLM/06-ChatGLM3-6B-Lora微调.md rename to models/ChatGLM/06-ChatGLM3-6B-Lora微调.md diff --git a/ChatGLM/06-ChatGLM3-6B-Lora微调.py b/models/ChatGLM/06-ChatGLM3-6B-Lora微调.py similarity index 100% rename from ChatGLM/06-ChatGLM3-6B-Lora微调.py rename to models/ChatGLM/06-ChatGLM3-6B-Lora微调.py diff --git a/ChatGLM/images/image-1.png b/models/ChatGLM/images/image-1.png similarity index 100% rename from ChatGLM/images/image-1.png rename to models/ChatGLM/images/image-1.png diff --git a/ChatGLM/images/image-2.png b/models/ChatGLM/images/image-2.png similarity index 100% rename from ChatGLM/images/image-2.png rename to models/ChatGLM/images/image-2.png diff --git a/ChatGLM/images/image-3.png b/models/ChatGLM/images/image-3.png similarity index 100% rename from ChatGLM/images/image-3.png rename to models/ChatGLM/images/image-3.png diff --git a/ChatGLM/images/image-4.png b/models/ChatGLM/images/image-4.png similarity index 100% rename from ChatGLM/images/image-4.png rename to models/ChatGLM/images/image-4.png diff --git a/ChatGLM/images/image-5.png b/models/ChatGLM/images/image-5.png similarity index 100% rename from ChatGLM/images/image-5.png rename to models/ChatGLM/images/image-5.png diff --git a/ChatGLM/images/image-6.png b/models/ChatGLM/images/image-6.png similarity index 100% rename from ChatGLM/images/image-6.png rename to models/ChatGLM/images/image-6.png diff --git a/ChatGLM/images/image-7.png b/models/ChatGLM/images/image-7.png similarity index 100% rename from ChatGLM/images/image-7.png rename to models/ChatGLM/images/image-7.png diff --git a/ChatGLM/images/image-8.png b/models/ChatGLM/images/image-8.png similarity index 100% rename from ChatGLM/images/image-8.png rename to models/ChatGLM/images/image-8.png diff --git a/ChatGLM/images/image-9.png b/models/ChatGLM/images/image-9.png similarity index 100% rename from ChatGLM/images/image-9.png rename to models/ChatGLM/images/image-9.png diff --git a/DeepSeek-Coder-V2/01-DeepSeek-Coder-V2-Lite-Instruct FastApi 部署调用.md b/models/DeepSeek-Coder-V2/01-DeepSeek-Coder-V2-Lite-Instruct FastApi 部署调用.md similarity index 100% rename from DeepSeek-Coder-V2/01-DeepSeek-Coder-V2-Lite-Instruct FastApi 部署调用.md rename to models/DeepSeek-Coder-V2/01-DeepSeek-Coder-V2-Lite-Instruct FastApi 部署调用.md diff --git a/DeepSeek-Coder-V2/02-DeepSeek-Coder-V2-Lite-Instruct 接入 LangChain.md b/models/DeepSeek-Coder-V2/02-DeepSeek-Coder-V2-Lite-Instruct 接入 LangChain.md similarity index 100% rename from DeepSeek-Coder-V2/02-DeepSeek-Coder-V2-Lite-Instruct 接入 LangChain.md rename to models/DeepSeek-Coder-V2/02-DeepSeek-Coder-V2-Lite-Instruct 接入 LangChain.md diff --git a/DeepSeek-Coder-V2/03-DeepSeek-Coder-V2-Lite-Instruct WebDemo 部署.md b/models/DeepSeek-Coder-V2/03-DeepSeek-Coder-V2-Lite-Instruct WebDemo 部署.md similarity index 100% rename from DeepSeek-Coder-V2/03-DeepSeek-Coder-V2-Lite-Instruct WebDemo 部署.md rename to models/DeepSeek-Coder-V2/03-DeepSeek-Coder-V2-Lite-Instruct WebDemo 部署.md diff --git a/DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.ipynb b/models/DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.ipynb similarity index 100% rename from DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.ipynb rename to models/DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.ipynb diff --git a/DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.md b/models/DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.md similarity index 100% rename from DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.md rename to models/DeepSeek-Coder-V2/04-DeepSeek-Coder-V2-Lite-Instruct Lora 微调.md diff --git a/DeepSeek-Coder-V2/images/fig1-1.png b/models/DeepSeek-Coder-V2/images/fig1-1.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-1.png rename to models/DeepSeek-Coder-V2/images/fig1-1.png diff --git a/DeepSeek-Coder-V2/images/fig1-2.png b/models/DeepSeek-Coder-V2/images/fig1-2.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-2.png rename to models/DeepSeek-Coder-V2/images/fig1-2.png diff --git a/DeepSeek-Coder-V2/images/fig1-3.png b/models/DeepSeek-Coder-V2/images/fig1-3.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-3.png rename to models/DeepSeek-Coder-V2/images/fig1-3.png diff --git a/DeepSeek-Coder-V2/images/fig1-4.png b/models/DeepSeek-Coder-V2/images/fig1-4.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-4.png rename to models/DeepSeek-Coder-V2/images/fig1-4.png diff --git a/DeepSeek-Coder-V2/images/fig1-5.png b/models/DeepSeek-Coder-V2/images/fig1-5.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-5.png rename to models/DeepSeek-Coder-V2/images/fig1-5.png diff --git a/DeepSeek-Coder-V2/images/fig1-6.png b/models/DeepSeek-Coder-V2/images/fig1-6.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-6.png rename to models/DeepSeek-Coder-V2/images/fig1-6.png diff --git a/DeepSeek-Coder-V2/images/fig1-7.png b/models/DeepSeek-Coder-V2/images/fig1-7.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-7.png rename to models/DeepSeek-Coder-V2/images/fig1-7.png diff --git a/DeepSeek-Coder-V2/images/fig1-8.png b/models/DeepSeek-Coder-V2/images/fig1-8.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-8.png rename to models/DeepSeek-Coder-V2/images/fig1-8.png diff --git a/DeepSeek-Coder-V2/images/fig1-9.png b/models/DeepSeek-Coder-V2/images/fig1-9.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig1-9.png rename to models/DeepSeek-Coder-V2/images/fig1-9.png diff --git a/DeepSeek-Coder-V2/images/fig2-1.png b/models/DeepSeek-Coder-V2/images/fig2-1.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig2-1.png rename to models/DeepSeek-Coder-V2/images/fig2-1.png diff --git a/DeepSeek-Coder-V2/images/fig2-2.png b/models/DeepSeek-Coder-V2/images/fig2-2.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig2-2.png rename to models/DeepSeek-Coder-V2/images/fig2-2.png diff --git a/DeepSeek-Coder-V2/images/fig2-3.png b/models/DeepSeek-Coder-V2/images/fig2-3.png similarity index 100% rename from DeepSeek-Coder-V2/images/fig2-3.png rename to models/DeepSeek-Coder-V2/images/fig2-3.png diff --git a/DeepSeek-Coder-V2/images/image03-1.png b/models/DeepSeek-Coder-V2/images/image03-1.png similarity index 100% rename from DeepSeek-Coder-V2/images/image03-1.png rename to models/DeepSeek-Coder-V2/images/image03-1.png diff --git a/DeepSeek-Coder-V2/images/image03-2.png b/models/DeepSeek-Coder-V2/images/image03-2.png similarity index 100% rename from DeepSeek-Coder-V2/images/image03-2.png rename to models/DeepSeek-Coder-V2/images/image03-2.png diff --git a/DeepSeek-Coder-V2/images/image03-3.png b/models/DeepSeek-Coder-V2/images/image03-3.png similarity index 100% rename from DeepSeek-Coder-V2/images/image03-3.png rename to models/DeepSeek-Coder-V2/images/image03-3.png diff --git a/DeepSeek-Coder-V2/images/image03-4.png b/models/DeepSeek-Coder-V2/images/image03-4.png similarity index 100% rename from DeepSeek-Coder-V2/images/image03-4.png rename to models/DeepSeek-Coder-V2/images/image03-4.png diff --git a/DeepSeek-Coder-V2/images/image03-5.png b/models/DeepSeek-Coder-V2/images/image03-5.png similarity index 100% rename from DeepSeek-Coder-V2/images/image03-5.png rename to models/DeepSeek-Coder-V2/images/image03-5.png diff --git a/DeepSeek-Coder-V2/images/image03-6.png b/models/DeepSeek-Coder-V2/images/image03-6.png similarity index 100% rename from DeepSeek-Coder-V2/images/image03-6.png rename to models/DeepSeek-Coder-V2/images/image03-6.png diff --git a/DeepSeek/01-DeepSeek-7B-chat FastApi.md b/models/DeepSeek/01-DeepSeek-7B-chat FastApi.md similarity index 100% rename from DeepSeek/01-DeepSeek-7B-chat FastApi.md rename to models/DeepSeek/01-DeepSeek-7B-chat FastApi.md diff --git a/DeepSeek/02-DeepSeek-7B-chat langchain.md b/models/DeepSeek/02-DeepSeek-7B-chat langchain.md similarity index 100% rename from DeepSeek/02-DeepSeek-7B-chat langchain.md rename to models/DeepSeek/02-DeepSeek-7B-chat langchain.md diff --git a/DeepSeek/03-DeepSeek-7B-chat WebDemo.md b/models/DeepSeek/03-DeepSeek-7B-chat WebDemo.md similarity index 100% rename from DeepSeek/03-DeepSeek-7B-chat WebDemo.md rename to models/DeepSeek/03-DeepSeek-7B-chat WebDemo.md diff --git a/DeepSeek/04-DeepSeek-7B-chat Lora 微调.ipynb b/models/DeepSeek/04-DeepSeek-7B-chat Lora 微调.ipynb similarity index 100% rename from DeepSeek/04-DeepSeek-7B-chat Lora 微调.ipynb rename to models/DeepSeek/04-DeepSeek-7B-chat Lora 微调.ipynb diff --git a/DeepSeek/04-DeepSeek-7B-chat Lora 微调.md b/models/DeepSeek/04-DeepSeek-7B-chat Lora 微调.md similarity index 100% rename from DeepSeek/04-DeepSeek-7B-chat Lora 微调.md rename to models/DeepSeek/04-DeepSeek-7B-chat Lora 微调.md diff --git a/DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.ipynb b/models/DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.ipynb similarity index 100% rename from DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.ipynb rename to models/DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.ipynb diff --git a/DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.md b/models/DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.md similarity index 100% rename from DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.md rename to models/DeepSeek/05-DeepSeek-7B-chat 4bits量化 Qlora 微调.md diff --git a/DeepSeek/06-DeepSeek-MoE-16b-chat FastApi部署调用.md b/models/DeepSeek/06-DeepSeek-MoE-16b-chat FastApi部署调用.md similarity index 100% rename from DeepSeek/06-DeepSeek-MoE-16b-chat FastApi部署调用.md rename to models/DeepSeek/06-DeepSeek-MoE-16b-chat FastApi部署调用.md diff --git a/DeepSeek/06-DeepSeek-MoE-16b-chat Transformer部署调用.md b/models/DeepSeek/06-DeepSeek-MoE-16b-chat Transformer部署调用.md similarity index 100% rename from DeepSeek/06-DeepSeek-MoE-16b-chat Transformer部署调用.md rename to models/DeepSeek/06-DeepSeek-MoE-16b-chat Transformer部署调用.md diff --git a/DeepSeek/07-deepseek_fine_tune.ipynb b/models/DeepSeek/07-deepseek_fine_tune.ipynb similarity index 100% rename from DeepSeek/07-deepseek_fine_tune.ipynb rename to models/DeepSeek/07-deepseek_fine_tune.ipynb diff --git a/DeepSeek/08-deepseek_web_demo.ipynb b/models/DeepSeek/08-deepseek_web_demo.ipynb similarity index 100% rename from DeepSeek/08-deepseek_web_demo.ipynb rename to models/DeepSeek/08-deepseek_web_demo.ipynb diff --git a/DeepSeek/images/image-1.png b/models/DeepSeek/images/image-1.png similarity index 100% rename from DeepSeek/images/image-1.png rename to models/DeepSeek/images/image-1.png diff --git a/DeepSeek/images/image-2.png b/models/DeepSeek/images/image-2.png similarity index 100% rename from DeepSeek/images/image-2.png rename to models/DeepSeek/images/image-2.png diff --git a/DeepSeek/images/image-3.png b/models/DeepSeek/images/image-3.png similarity index 100% rename from DeepSeek/images/image-3.png rename to models/DeepSeek/images/image-3.png diff --git a/DeepSeek/images/image-4.png b/models/DeepSeek/images/image-4.png similarity index 100% rename from DeepSeek/images/image-4.png rename to models/DeepSeek/images/image-4.png diff --git a/DeepSeek/images/image-5.png b/models/DeepSeek/images/image-5.png similarity index 100% rename from DeepSeek/images/image-5.png rename to models/DeepSeek/images/image-5.png diff --git a/DeepSeek/images/image-6.png b/models/DeepSeek/images/image-6.png similarity index 100% rename from DeepSeek/images/image-6.png rename to models/DeepSeek/images/image-6.png diff --git a/DeepSeek/images/image-7.png b/models/DeepSeek/images/image-7.png similarity index 100% rename from DeepSeek/images/image-7.png rename to models/DeepSeek/images/image-7.png diff --git a/DeepSeek/images/image-8.png b/models/DeepSeek/images/image-8.png similarity index 100% rename from DeepSeek/images/image-8.png rename to models/DeepSeek/images/image-8.png diff --git a/DeepSeek/images/image-9.png b/models/DeepSeek/images/image-9.png similarity index 100% rename from DeepSeek/images/image-9.png rename to models/DeepSeek/images/image-9.png diff --git a/GLM-4/01-GLM-4-9B-chat FastApi 部署调用.md b/models/GLM-4/01-GLM-4-9B-chat FastApi 部署调用.md similarity index 100% rename from GLM-4/01-GLM-4-9B-chat FastApi 部署调用.md rename to models/GLM-4/01-GLM-4-9B-chat FastApi 部署调用.md diff --git a/GLM-4/02-GLM-4-9B-chat langchain 接入.md b/models/GLM-4/02-GLM-4-9B-chat langchain 接入.md similarity index 100% rename from GLM-4/02-GLM-4-9B-chat langchain 接入.md rename to models/GLM-4/02-GLM-4-9B-chat langchain 接入.md diff --git a/GLM-4/03-GLM-4-9B-Chat WebDemo.md b/models/GLM-4/03-GLM-4-9B-Chat WebDemo.md similarity index 100% rename from GLM-4/03-GLM-4-9B-Chat WebDemo.md rename to models/GLM-4/03-GLM-4-9B-Chat WebDemo.md diff --git a/GLM-4/04-GLM-4-9B-Chat vLLM 部署调用.md b/models/GLM-4/04-GLM-4-9B-Chat vLLM 部署调用.md similarity index 100% rename from GLM-4/04-GLM-4-9B-Chat vLLM 部署调用.md rename to models/GLM-4/04-GLM-4-9B-Chat vLLM 部署调用.md diff --git a/GLM-4/05-GLM-4-9B-chat Lora 微调.ipynb b/models/GLM-4/05-GLM-4-9B-chat Lora 微调.ipynb similarity index 100% rename from GLM-4/05-GLM-4-9B-chat Lora 微调.ipynb rename to models/GLM-4/05-GLM-4-9B-chat Lora 微调.ipynb diff --git a/GLM-4/05-GLM-4-9B-chat Lora 微调.md b/models/GLM-4/05-GLM-4-9B-chat Lora 微调.md similarity index 100% rename from GLM-4/05-GLM-4-9B-chat Lora 微调.md rename to models/GLM-4/05-GLM-4-9B-chat Lora 微调.md diff --git a/GLM-4/benchmark_throughput.py b/models/GLM-4/benchmark_throughput.py similarity index 100% rename from GLM-4/benchmark_throughput.py rename to models/GLM-4/benchmark_throughput.py diff --git a/GLM-4/images/image-1.png b/models/GLM-4/images/image-1.png similarity index 100% rename from GLM-4/images/image-1.png rename to models/GLM-4/images/image-1.png diff --git a/GLM-4/images/image01-1.png b/models/GLM-4/images/image01-1.png similarity index 100% rename from GLM-4/images/image01-1.png rename to models/GLM-4/images/image01-1.png diff --git a/GLM-4/images/image01-2.png b/models/GLM-4/images/image01-2.png similarity index 100% rename from GLM-4/images/image01-2.png rename to models/GLM-4/images/image01-2.png diff --git a/GLM-4/images/image01-3.png b/models/GLM-4/images/image01-3.png similarity index 100% rename from GLM-4/images/image01-3.png rename to models/GLM-4/images/image01-3.png diff --git a/GLM-4/images/image01-4.png b/models/GLM-4/images/image01-4.png similarity index 100% rename from GLM-4/images/image01-4.png rename to models/GLM-4/images/image01-4.png diff --git a/GLM-4/images/image01-5.png b/models/GLM-4/images/image01-5.png similarity index 100% rename from GLM-4/images/image01-5.png rename to models/GLM-4/images/image01-5.png diff --git a/GLM-4/images/image02-1.png b/models/GLM-4/images/image02-1.png similarity index 100% rename from GLM-4/images/image02-1.png rename to models/GLM-4/images/image02-1.png diff --git a/GLM-4/images/image03-1.png b/models/GLM-4/images/image03-1.png similarity index 100% rename from GLM-4/images/image03-1.png rename to models/GLM-4/images/image03-1.png diff --git a/GLM-4/images/image03-2.png b/models/GLM-4/images/image03-2.png similarity index 100% rename from GLM-4/images/image03-2.png rename to models/GLM-4/images/image03-2.png diff --git a/GLM-4/images/image04-1.png b/models/GLM-4/images/image04-1.png similarity index 100% rename from GLM-4/images/image04-1.png rename to models/GLM-4/images/image04-1.png diff --git a/Gemma/01-Gemma-2B-Instruct FastApi 部署调用.md b/models/Gemma/01-Gemma-2B-Instruct FastApi 部署调用.md similarity index 100% rename from Gemma/01-Gemma-2B-Instruct FastApi 部署调用.md rename to models/Gemma/01-Gemma-2B-Instruct FastApi 部署调用.md diff --git a/Gemma/02-Gemma-2B-Instruct langchain 接入.md b/models/Gemma/02-Gemma-2B-Instruct langchain 接入.md similarity index 100% rename from Gemma/02-Gemma-2B-Instruct langchain 接入.md rename to models/Gemma/02-Gemma-2B-Instruct langchain 接入.md diff --git a/Gemma/03-Gemma-2B-Instruct WebDemo 部署.md b/models/Gemma/03-Gemma-2B-Instruct WebDemo 部署.md similarity index 100% rename from Gemma/03-Gemma-2B-Instruct WebDemo 部署.md rename to models/Gemma/03-Gemma-2B-Instruct WebDemo 部署.md diff --git a/Gemma/04-Gemma-2B-Instruct Lora微调.md b/models/Gemma/04-Gemma-2B-Instruct Lora微调.md similarity index 100% rename from Gemma/04-Gemma-2B-Instruct Lora微调.md rename to models/Gemma/04-Gemma-2B-Instruct Lora微调.md diff --git a/Gemma/04-Gemma-2B-Lora微调.ipynb b/models/Gemma/04-Gemma-2B-Lora微调.ipynb similarity index 100% rename from Gemma/04-Gemma-2B-Lora微调.ipynb rename to models/Gemma/04-Gemma-2B-Lora微调.ipynb diff --git a/Gemma/images/image-1.png b/models/Gemma/images/image-1.png similarity index 100% rename from Gemma/images/image-1.png rename to models/Gemma/images/image-1.png diff --git a/Gemma/images/image-2.png b/models/Gemma/images/image-2.png similarity index 100% rename from Gemma/images/image-2.png rename to models/Gemma/images/image-2.png diff --git a/Gemma/images/image-3.png b/models/Gemma/images/image-3.png similarity index 100% rename from Gemma/images/image-3.png rename to models/Gemma/images/image-3.png diff --git a/Gemma/images/image-4.png b/models/Gemma/images/image-4.png similarity index 100% rename from Gemma/images/image-4.png rename to models/Gemma/images/image-4.png diff --git a/Gemma/images/image-5.png b/models/Gemma/images/image-5.png similarity index 100% rename from Gemma/images/image-5.png rename to models/Gemma/images/image-5.png diff --git a/Gemma2/01-Gemma-2-9b-it FastApi 部署调用.md b/models/Gemma2/01-Gemma-2-9b-it FastApi 部署调用.md similarity index 100% rename from Gemma2/01-Gemma-2-9b-it FastApi 部署调用.md rename to models/Gemma2/01-Gemma-2-9b-it FastApi 部署调用.md diff --git a/Gemma2/02-Gemma-2-9b-it langchain 接入.md b/models/Gemma2/02-Gemma-2-9b-it langchain 接入.md similarity index 100% rename from Gemma2/02-Gemma-2-9b-it langchain 接入.md rename to models/Gemma2/02-Gemma-2-9b-it langchain 接入.md diff --git a/Gemma2/03-Gemma-2-9b-it WebDemo 部署.md b/models/Gemma2/03-Gemma-2-9b-it WebDemo 部署.md similarity index 100% rename from Gemma2/03-Gemma-2-9b-it WebDemo 部署.md rename to models/Gemma2/03-Gemma-2-9b-it WebDemo 部署.md diff --git a/Gemma2/04-Gemma-2-9b-it peft lora微调.ipynb b/models/Gemma2/04-Gemma-2-9b-it peft lora微调.ipynb similarity index 100% rename from Gemma2/04-Gemma-2-9b-it peft lora微调.ipynb rename to models/Gemma2/04-Gemma-2-9b-it peft lora微调.ipynb diff --git a/Gemma2/04-Gemma-2-9b-it peft lora微调.md b/models/Gemma2/04-Gemma-2-9b-it peft lora微调.md similarity index 100% rename from Gemma2/04-Gemma-2-9b-it peft lora微调.md rename to models/Gemma2/04-Gemma-2-9b-it peft lora微调.md diff --git a/Gemma2/images/01-1.png b/models/Gemma2/images/01-1.png similarity index 100% rename from Gemma2/images/01-1.png rename to models/Gemma2/images/01-1.png diff --git a/Gemma2/images/01-4-0.png b/models/Gemma2/images/01-4-0.png similarity index 100% rename from Gemma2/images/01-4-0.png rename to models/Gemma2/images/01-4-0.png diff --git a/Gemma2/images/01-4-1.png b/models/Gemma2/images/01-4-1.png similarity index 100% rename from Gemma2/images/01-4-1.png rename to models/Gemma2/images/01-4-1.png diff --git a/Gemma2/images/01-5.png b/models/Gemma2/images/01-5.png similarity index 100% rename from Gemma2/images/01-5.png rename to models/Gemma2/images/01-5.png diff --git a/Gemma2/images/01-6.png b/models/Gemma2/images/01-6.png similarity index 100% rename from Gemma2/images/01-6.png rename to models/Gemma2/images/01-6.png diff --git a/Gemma2/images/01-7.png b/models/Gemma2/images/01-7.png similarity index 100% rename from Gemma2/images/01-7.png rename to models/Gemma2/images/01-7.png diff --git a/Gemma2/images/02-1.png b/models/Gemma2/images/02-1.png similarity index 100% rename from Gemma2/images/02-1.png rename to models/Gemma2/images/02-1.png diff --git a/Gemma2/images/03-0.png b/models/Gemma2/images/03-0.png similarity index 100% rename from Gemma2/images/03-0.png rename to models/Gemma2/images/03-0.png diff --git a/Gemma2/images/03-1.png b/models/Gemma2/images/03-1.png similarity index 100% rename from Gemma2/images/03-1.png rename to models/Gemma2/images/03-1.png diff --git a/Gemma2/images/03-2.png b/models/Gemma2/images/03-2.png similarity index 100% rename from Gemma2/images/03-2.png rename to models/Gemma2/images/03-2.png diff --git a/Gemma2/images/03-3.png b/models/Gemma2/images/03-3.png similarity index 100% rename from Gemma2/images/03-3.png rename to models/Gemma2/images/03-3.png diff --git a/Gemma2/images/04-1.png b/models/Gemma2/images/04-1.png similarity index 100% rename from Gemma2/images/04-1.png rename to models/Gemma2/images/04-1.png diff --git a/Gemma2/images/04-2.png b/models/Gemma2/images/04-2.png similarity index 100% rename from Gemma2/images/04-2.png rename to models/Gemma2/images/04-2.png diff --git a/General-Setting/01-pip、conda换源.md b/models/General-Setting/01-pip、conda换源.md similarity index 100% rename from General-Setting/01-pip、conda换源.md rename to models/General-Setting/01-pip、conda换源.md diff --git a/General-Setting/02-AutoDL开放端口.md b/models/General-Setting/02-AutoDL开放端口.md similarity index 100% rename from General-Setting/02-AutoDL开放端口.md rename to models/General-Setting/02-AutoDL开放端口.md diff --git a/General-Setting/03-模型下载.md b/models/General-Setting/03-模型下载.md similarity index 100% rename from General-Setting/03-模型下载.md rename to models/General-Setting/03-模型下载.md diff --git a/General-Setting/04-Issue&PR&update.md b/models/General-Setting/04-Issue&PR&update.md similarity index 100% rename from General-Setting/04-Issue&PR&update.md rename to models/General-Setting/04-Issue&PR&update.md diff --git a/General-Setting/pic/Issue1.png b/models/General-Setting/pic/Issue1.png similarity index 100% rename from General-Setting/pic/Issue1.png rename to models/General-Setting/pic/Issue1.png diff --git a/General-Setting/pic/Issue2.png b/models/General-Setting/pic/Issue2.png similarity index 100% rename from General-Setting/pic/Issue2.png rename to models/General-Setting/pic/Issue2.png diff --git a/General-Setting/pic/PR.png b/models/General-Setting/pic/PR.png similarity index 100% rename from General-Setting/pic/PR.png rename to models/General-Setting/pic/PR.png diff --git a/General-Setting/pic/PR1.png b/models/General-Setting/pic/PR1.png similarity index 100% rename from General-Setting/pic/PR1.png rename to models/General-Setting/pic/PR1.png diff --git a/General-Setting/pic/PR3.png b/models/General-Setting/pic/PR3.png similarity index 100% rename from General-Setting/pic/PR3.png rename to models/General-Setting/pic/PR3.png diff --git a/General-Setting/pic/PR4.png b/models/General-Setting/pic/PR4.png similarity index 100% rename from General-Setting/pic/PR4.png rename to models/General-Setting/pic/PR4.png diff --git a/General-Setting/pic/PR5.png b/models/General-Setting/pic/PR5.png similarity index 100% rename from General-Setting/pic/PR5.png rename to models/General-Setting/pic/PR5.png diff --git a/General-Setting/pic/PR6.png b/models/General-Setting/pic/PR6.png similarity index 100% rename from General-Setting/pic/PR6.png rename to models/General-Setting/pic/PR6.png diff --git a/General-Setting/pic/端口映射.png b/models/General-Setting/pic/端口映射.png similarity index 100% rename from General-Setting/pic/端口映射.png rename to models/General-Setting/pic/端口映射.png diff --git a/InternLM/01-InternLM-Chat-7B Transformers 部署调用.md b/models/InternLM/01-InternLM-Chat-7B Transformers 部署调用.md similarity index 100% rename from InternLM/01-InternLM-Chat-7B Transformers 部署调用.md rename to models/InternLM/01-InternLM-Chat-7B Transformers 部署调用.md diff --git a/InternLM/02-internLM-Chat-7B FastApi.md b/models/InternLM/02-internLM-Chat-7B FastApi.md similarity index 100% rename from InternLM/02-internLM-Chat-7B FastApi.md rename to models/InternLM/02-internLM-Chat-7B FastApi.md diff --git a/InternLM/03-InternLM-Chat-7B.md b/models/InternLM/03-InternLM-Chat-7B.md similarity index 100% rename from InternLM/03-InternLM-Chat-7B.md rename to models/InternLM/03-InternLM-Chat-7B.md diff --git a/InternLM/04-Lagent+InternLM-Chat-7B-V1.1.md b/models/InternLM/04-Lagent+InternLM-Chat-7B-V1.1.md similarity index 100% rename from InternLM/04-Lagent+InternLM-Chat-7B-V1.1.md rename to models/InternLM/04-Lagent+InternLM-Chat-7B-V1.1.md diff --git a/InternLM/05-浦语灵笔图文理解&创作.md b/models/InternLM/05-浦语灵笔图文理解&创作.md similarity index 100% rename from InternLM/05-浦语灵笔图文理解&创作.md rename to models/InternLM/05-浦语灵笔图文理解&创作.md diff --git a/InternLM/06-InternLM接入LangChain搭建知识库助手.md b/models/InternLM/06-InternLM接入LangChain搭建知识库助手.md similarity index 100% rename from InternLM/06-InternLM接入LangChain搭建知识库助手.md rename to models/InternLM/06-InternLM接入LangChain搭建知识库助手.md diff --git a/InternLM/06-InternLM接入LangChain搭建知识库助手/LLM.py b/models/InternLM/06-InternLM接入LangChain搭建知识库助手/LLM.py similarity index 100% rename from InternLM/06-InternLM接入LangChain搭建知识库助手/LLM.py rename to models/InternLM/06-InternLM接入LangChain搭建知识库助手/LLM.py diff --git a/InternLM/06-InternLM接入LangChain搭建知识库助手/creat_db.py b/models/InternLM/06-InternLM接入LangChain搭建知识库助手/creat_db.py similarity index 100% rename from InternLM/06-InternLM接入LangChain搭建知识库助手/creat_db.py rename to models/InternLM/06-InternLM接入LangChain搭建知识库助手/creat_db.py diff --git a/InternLM/06-InternLM接入LangChain搭建知识库助手/readme.md b/models/InternLM/06-InternLM接入LangChain搭建知识库助手/readme.md similarity index 100% rename from InternLM/06-InternLM接入LangChain搭建知识库助手/readme.md rename to models/InternLM/06-InternLM接入LangChain搭建知识库助手/readme.md diff --git a/InternLM/06-InternLM接入LangChain搭建知识库助手/run_gradio.py b/models/InternLM/06-InternLM接入LangChain搭建知识库助手/run_gradio.py similarity index 100% rename from InternLM/06-InternLM接入LangChain搭建知识库助手/run_gradio.py rename to models/InternLM/06-InternLM接入LangChain搭建知识库助手/run_gradio.py diff --git a/InternLM/images/image-1.png b/models/InternLM/images/image-1.png similarity index 100% rename from InternLM/images/image-1.png rename to models/InternLM/images/image-1.png diff --git a/InternLM/images/image-10.png b/models/InternLM/images/image-10.png similarity index 100% rename from InternLM/images/image-10.png rename to models/InternLM/images/image-10.png diff --git a/InternLM/images/image-11.png b/models/InternLM/images/image-11.png similarity index 100% rename from InternLM/images/image-11.png rename to models/InternLM/images/image-11.png diff --git a/InternLM/images/image-12.png b/models/InternLM/images/image-12.png similarity index 100% rename from InternLM/images/image-12.png rename to models/InternLM/images/image-12.png diff --git a/InternLM/images/image-13.png b/models/InternLM/images/image-13.png similarity index 100% rename from InternLM/images/image-13.png rename to models/InternLM/images/image-13.png diff --git a/InternLM/images/image-14.png b/models/InternLM/images/image-14.png similarity index 100% rename from InternLM/images/image-14.png rename to models/InternLM/images/image-14.png diff --git a/InternLM/images/image-2.png b/models/InternLM/images/image-2.png similarity index 100% rename from InternLM/images/image-2.png rename to models/InternLM/images/image-2.png diff --git a/InternLM/images/image-3.png b/models/InternLM/images/image-3.png similarity index 100% rename from InternLM/images/image-3.png rename to models/InternLM/images/image-3.png diff --git a/InternLM/images/image-4.png b/models/InternLM/images/image-4.png similarity index 100% rename from InternLM/images/image-4.png rename to models/InternLM/images/image-4.png diff --git a/InternLM/images/image-5.png b/models/InternLM/images/image-5.png similarity index 100% rename from InternLM/images/image-5.png rename to models/InternLM/images/image-5.png diff --git a/InternLM/images/image-6.png b/models/InternLM/images/image-6.png similarity index 100% rename from InternLM/images/image-6.png rename to models/InternLM/images/image-6.png diff --git a/InternLM/images/image-7.png b/models/InternLM/images/image-7.png similarity index 100% rename from InternLM/images/image-7.png rename to models/InternLM/images/image-7.png diff --git a/InternLM/images/image-8.png b/models/InternLM/images/image-8.png similarity index 100% rename from InternLM/images/image-8.png rename to models/InternLM/images/image-8.png diff --git a/InternLM/images/image-9.png b/models/InternLM/images/image-9.png similarity index 100% rename from InternLM/images/image-9.png rename to models/InternLM/images/image-9.png diff --git a/InternLM/images/image.png b/models/InternLM/images/image.png similarity index 100% rename from InternLM/images/image.png rename to models/InternLM/images/image.png diff --git a/InternLM2/01-InternLM2-7B-chat FastAPI部署.md b/models/InternLM2/01-InternLM2-7B-chat FastAPI部署.md similarity index 100% rename from InternLM2/01-InternLM2-7B-chat FastAPI部署.md rename to models/InternLM2/01-InternLM2-7B-chat FastAPI部署.md diff --git a/InternLM2/02-InternLM2-7B-chat langchain 接入.md b/models/InternLM2/02-InternLM2-7B-chat langchain 接入.md similarity index 100% rename from InternLM2/02-InternLM2-7B-chat langchain 接入.md rename to models/InternLM2/02-InternLM2-7B-chat langchain 接入.md diff --git a/InternLM2/03-InternLM2-7B-chat WebDemo 部署.md b/models/InternLM2/03-InternLM2-7B-chat WebDemo 部署.md similarity index 97% rename from InternLM2/03-InternLM2-7B-chat WebDemo 部署.md rename to models/InternLM2/03-InternLM2-7B-chat WebDemo 部署.md index fbb0bf8..24c74ba 100644 --- a/InternLM2/03-InternLM2-7B-chat WebDemo 部署.md +++ b/models/InternLM2/03-InternLM2-7B-chat WebDemo 部署.md @@ -1,131 +1,131 @@ -# InternLM2-7B-chat WebDemo 部署 - -InternLM2 ,即书生·浦语大模型第二代,开源了面向实用场景的70亿参数基础模型与对话模型 (InternLM2-Chat-7B)。模型具有以下特点: - -- 有效支持20万字超长上下文:模型在20万字长输入中几乎完美地实现长文“大海捞针”,而且在 LongBench 和 L-Eval 等长文任务中的表现也达到开源模型中的领先水平。 可以通过 LMDeploy 尝试20万字超长上下文推理。 -- 综合性能全面提升:各能力维度相比上一代模型全面进步,在推理、数学、代码、对话体验、指令遵循和创意写作等方面的能力提升尤为显著,综合性能达到同量级开源模型的领先水平,在重点能力评测上 InternLM2-Chat-20B 能比肩甚至超越 ChatGPT (GPT-3.5)。 -- 代码解释器与数据分析:在配合代码解释器(code-interpreter)的条件下,InternLM2-Chat-20B 在 GSM8K 和 MATH 上可以达到和 GPT-4 相仿的水平。基于在数理和工具方面强大的基础能力,InternLM2-Chat 提供了实用的数据分析能力。 -- 工具调用能力整体升级:基于更强和更具有泛化性的指令理解、工具筛选与结果反思等能力,新版模型可以更可靠地支持复杂智能体的搭建,支持对工具进行有效的多轮调用,完成较复杂的任务。 - -## 环境准备 - -在 Autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8。 - -![3-1](images/3-1.png) - -接下来打开刚刚租用服务器的 JupyterLab,新建一个`Internlm2-7b-chat-web.ipynb`文件 - -![3-2](images/3-2.png) - -pip换源和安装依赖包,在ipynb文件里写入下面代码,点击运行 - -``` -# 升级pip -!python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -!pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -# 安装python依赖 -!pip install modelscope==1.9.5 -!pip install transformers==4.36.2 -!pip install streamlit==1.24.0 -!pip install sentencepiece==0.1.99 -!pip install accelerate==0.24.1 -!pip install transformers_stream_generator==0.0.4 -pip install protobuf -``` - -如果你是在终端命令运行直接就按下面的命令运行 - -```bash -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -# 安装python依赖 -pip install modelscope==1.9.5 -pip install transformers==4.36.2 -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.24.1 -pip install transformers_stream_generator==0.0.4 -``` - -## 模型下载 - -InternLM2-chat-7b 模型: - -* [huggingface](https://huggingface.co/internlm/internlm2-chat-7b) -* [modelscope](https://modelscope.cn/models/Shanghai_AI_Laboratory/internlm2-chat-7b/summary) - -### 使用modelscope下载 - -使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 - -在`Internlm2-7b-chat-web.ipynb`文件中新建一个代码块,运行下载`internlm2-chat-7b`模型。模型下载需要时间,我们直接往下看[代码准备](#代码准备) - -``` -from modelscope import snapshot_download - -model_dir = snapshot_download('Shanghai_AI_Laboratory/internlm2-chat-7b', cache_dir='/root/autodl-tmp', revision='master') -``` - -![3-3](images/3-3.png) - -## 代码准备 - -### 源码拉取 - -以下操作,可以在jupyter运行下载模型的过程中,你新开一个命令行终端进行操作 - -``` -# 启动镜像加速 -source /etc/network_turbo - -cd /root/autodl-tmp -# 下载 Internlm 代码 -git clone https://github.com/InternLM/InternLM.git -# 取消代理 -unset http_proxy && unset https_proxy -``` - -![3-4](images/3-4.png) - -### 安装依赖 - -``` -# 进入源码目录 -cd /root/autodl-tmp/InternLM/ -# 安装internlm依赖 -pip install -r requirements.txt -``` - -### 使用**InternLM**的web_demo运行 - -将 `/root/autodl-tmp/InternLM/chat/web_demo.py`中 183 行和 186 行的模型更换为本地的`/root/autodl-tmp/Shanghai_AI_Laboratory/internlm2-chat-7b`。 - -![3-5](images/3-5.png) - -修改完成之后,启动`web_demo.py`文件 - -``` -# 进入源码目录 -cd /root/autodl-tmp/InternLM/ -streamlit run ./chat/web_demo.py -``` - -![3-6](images/3-6.png) - -此时,我们通过ssh端口转发,把`autodl`上启动的服务映射到本地端口上来,使用下面的命令。在本地打开`powershell` - -``` -ssh -CNg -L 8501:127.0.0.1:8501 -p 【你的autodl机器的ssh端口】 root@[你的autodl机器地址] -ssh -CNg -L 8501:127.0.0.1:8501 -p 36494 root@region-45.autodl.pro -``` - -![3-7](images/3-7.png) - -在加载完模型之后,就可以既可与InternLM2-Chat-7B进行对话了,如下图所示: - +# InternLM2-7B-chat WebDemo 部署 + +InternLM2 ,即书生·浦语大模型第二代,开源了面向实用场景的70亿参数基础模型与对话模型 (InternLM2-Chat-7B)。模型具有以下特点: + +- 有效支持20万字超长上下文:模型在20万字长输入中几乎完美地实现长文“大海捞针”,而且在 LongBench 和 L-Eval 等长文任务中的表现也达到开源模型中的领先水平。 可以通过 LMDeploy 尝试20万字超长上下文推理。 +- 综合性能全面提升:各能力维度相比上一代模型全面进步,在推理、数学、代码、对话体验、指令遵循和创意写作等方面的能力提升尤为显著,综合性能达到同量级开源模型的领先水平,在重点能力评测上 InternLM2-Chat-20B 能比肩甚至超越 ChatGPT (GPT-3.5)。 +- 代码解释器与数据分析:在配合代码解释器(code-interpreter)的条件下,InternLM2-Chat-20B 在 GSM8K 和 MATH 上可以达到和 GPT-4 相仿的水平。基于在数理和工具方面强大的基础能力,InternLM2-Chat 提供了实用的数据分析能力。 +- 工具调用能力整体升级:基于更强和更具有泛化性的指令理解、工具筛选与结果反思等能力,新版模型可以更可靠地支持复杂智能体的搭建,支持对工具进行有效的多轮调用,完成较复杂的任务。 + +## 环境准备 + +在 Autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8。 + +![3-1](images/3-1.png) + +接下来打开刚刚租用服务器的 JupyterLab,新建一个`Internlm2-7b-chat-web.ipynb`文件 + +![3-2](images/3-2.png) + +pip换源和安装依赖包,在ipynb文件里写入下面代码,点击运行 + +``` +# 升级pip +!python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +!pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +# 安装python依赖 +!pip install modelscope==1.9.5 +!pip install transformers==4.36.2 +!pip install streamlit==1.24.0 +!pip install sentencepiece==0.1.99 +!pip install accelerate==0.24.1 +!pip install transformers_stream_generator==0.0.4 +pip install protobuf +``` + +如果你是在终端命令运行直接就按下面的命令运行 + +```bash +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +# 安装python依赖 +pip install modelscope==1.9.5 +pip install transformers==4.36.2 +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.24.1 +pip install transformers_stream_generator==0.0.4 +``` + +## 模型下载 + +InternLM2-chat-7b 模型: + +* [huggingface](https://huggingface.co/internlm/internlm2-chat-7b) +* [modelscope](https://modelscope.cn/models/Shanghai_AI_Laboratory/internlm2-chat-7b/summary) + +### 使用modelscope下载 + +使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 + +在`Internlm2-7b-chat-web.ipynb`文件中新建一个代码块,运行下载`internlm2-chat-7b`模型。模型下载需要时间,我们直接往下看[代码准备](#代码准备) + +``` +from modelscope import snapshot_download + +model_dir = snapshot_download('Shanghai_AI_Laboratory/internlm2-chat-7b', cache_dir='/root/autodl-tmp', revision='master') +``` + +![3-3](images/3-3.png) + +## 代码准备 + +### 源码拉取 + +以下操作,可以在jupyter运行下载模型的过程中,你新开一个命令行终端进行操作 + +``` +# 启动镜像加速 +source /etc/network_turbo + +cd /root/autodl-tmp +# 下载 Internlm 代码 +git clone https://github.com/InternLM/InternLM.git +# 取消代理 +unset http_proxy && unset https_proxy +``` + +![3-4](images/3-4.png) + +### 安装依赖 + +``` +# 进入源码目录 +cd /root/autodl-tmp/InternLM/ +# 安装internlm依赖 +pip install -r requirements.txt +``` + +### 使用**InternLM**的web_demo运行 + +将 `/root/autodl-tmp/InternLM/chat/web_demo.py`中 183 行和 186 行的模型更换为本地的`/root/autodl-tmp/Shanghai_AI_Laboratory/internlm2-chat-7b`。 + +![3-5](images/3-5.png) + +修改完成之后,启动`web_demo.py`文件 + +``` +# 进入源码目录 +cd /root/autodl-tmp/InternLM/ +streamlit run ./chat/web_demo.py +``` + +![3-6](images/3-6.png) + +此时,我们通过ssh端口转发,把`autodl`上启动的服务映射到本地端口上来,使用下面的命令。在本地打开`powershell` + +``` +ssh -CNg -L 8501:127.0.0.1:8501 -p 【你的autodl机器的ssh端口】 root@[你的autodl机器地址] +ssh -CNg -L 8501:127.0.0.1:8501 -p 36494 root@region-45.autodl.pro +``` + +![3-7](images/3-7.png) + +在加载完模型之后,就可以既可与InternLM2-Chat-7B进行对话了,如下图所示: + ![3-8](images/3-8.png) \ No newline at end of file diff --git a/InternLM2/04-InternLM2-7B-chat Xtuner Qlora 微调.md b/models/InternLM2/04-InternLM2-7B-chat Xtuner Qlora 微调.md similarity index 97% rename from InternLM2/04-InternLM2-7B-chat Xtuner Qlora 微调.md rename to models/InternLM2/04-InternLM2-7B-chat Xtuner Qlora 微调.md index 40c3f74..cc47bfc 100644 --- a/InternLM2/04-InternLM2-7B-chat Xtuner Qlora 微调.md +++ b/models/InternLM2/04-InternLM2-7B-chat Xtuner Qlora 微调.md @@ -1,311 +1,311 @@ -# InternLM2-7B-chat Xtuner Qlora 微调 - -## Xtuner介绍 -
- -

-
- -XTuner是上海人工智能实验室开发的低成本大模型训练工具箱,XTuner 是一个高效、灵活、全能的轻量化大模型微调工具库。只要**8G**。最低只需 **8GB** 显存,就可以微调InternLM2-7B模型,打造专属于你的 AI 助手。 - -仓库地址:https://github.com/InternLM/xtuner - -### Xtuner特点 - -**高效** - -- 支持大语言模型 LLM、多模态图文模型 VLM 的预训练及轻量级微调。XTuner 支持在 8GB 显存下微调 7B 模型,同时也支持多节点跨设备微调更大尺度模型(70B+)。 -- 自动分发高性能算子(如 FlashAttention、Triton kernels 等)以加速训练吞吐。 -- 兼容 [DeepSpeed](https://github.com/microsoft/DeepSpeed) 🚀,轻松应用各种 ZeRO 训练优化策略。 - -**灵活** - -- 支持多种大语言模型,包括但不限于 [InternLM](https://huggingface.co/internlm)、[Mixtral-8x7B](https://huggingface.co/mistralai)、[Llama2](https://huggingface.co/meta-llama)、[ChatGLM](https://huggingface.co/THUDM)、[Qwen](https://huggingface.co/Qwen)、[Baichuan](https://huggingface.co/baichuan-inc)。 -- 支持多模态图文模型 LLaVA 的预训练与微调。利用 XTuner 训得模型 [LLaVA-InternLM2-20B](https://huggingface.co/xtuner/llava-internlm2-20b) 表现优异。 -- 精心设计的数据管道,兼容任意数据格式,开源数据或自定义数据皆可快速上手。 -- 支持 [QLoRA](http://arxiv.org/abs/2305.14314)、[LoRA](http://arxiv.org/abs/2106.09685)、全量参数微调等多种微调算法,支撑用户根据具体需求作出最优选择。 - -**全能** - -- 支持增量预训练、指令微调与 Agent 微调。 -- 预定义众多开源对话模版,支持与开源或训练所得模型进行对话。 -- 训练所得模型可无缝接入部署工具库 [LMDeploy](https://github.com/InternLM/lmdeploy)、大规模评测工具库 [OpenCompass](https://github.com/open-compass/opencompass) 及 [VLMEvalKit](https://github.com/open-compass/VLMEvalKit)。 - -## 环境准备 - -在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 。 - -接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 - -![机器配置选择](./images/1.png) - -### 创建工作目录 - -创建本次微调实践的工作目录`/root/autodl-tmp/ft-learn` - -``` -# 创建微调工作目录 -mkdir -p /root/autodl-tmp/ft-learn - -# 创建微调数据集存放目录 -mkdir -p /root/autodl-tmp/ft-learn/dataset - -# 创建微调配置文件存放目录 -mkdir -p /root/autodl-tmp/ft-learn/config - -``` - -### 安装依赖 - -```bash -# 升级pip -python -m pip install --upgrade pip -# 安装python依赖 -pip install modelscope==1.9.5 -pip install transformers==4.36.2 -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.24.1 -pip install transformers_stream_generator==0.0.4 -pip install einops ujson -pip install protobuf -``` - -### 使用modelscope下载模型 - -使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 - -在`Internlm2-7b-chat-web.ipynb`文件中新建一个代码块,运行下载`internlm2-chat-7b`模型。模型下载需要时间,我们直接往下看 - -``` -from modelscope import snapshot_download - -model_dir = snapshot_download('Shanghai_AI_Laboratory/internlm2-chat-7b', cache_dir='/root/autodl-tmp', revision='master') -``` - -![3-3](images/3-3.png) - -### 安装Xtuner - -```bash -git clone -b v0.1.14 https://github.com/InternLM/xtuner -cd xtuner -# 从源码安装 XTuner -pip install -e '.[all]' -# 安装完成之后就可以在命令行使用xtuner了 -# 查看xtuner使用帮助 -xtuner help -# 查看xtuner版本 -xtuner version -``` - -![4-1](images/4-1.png) - -## 数据集处理 - -我自己整理的`心理大模型-职场焦虑语料.xlsx`,通过`gen_qa_json.py`文件生成一个`career_coach.jsonl`文件 - -运行`python /root/autodl-tmp/ft-learn/dataset/gen_qa_json.py`生成文件,你们也可以按照我的数据语料格式,自定义你们自己的数据集。`gen_qa_json.py`文件代码如下: - -``` -import pandas as pd -import json - -# 读取Excel文件 -excel_file = './心理大模型-职场焦虑语料.xlsx' # 替换成实际的Excel文件路径 -df = pd.read_excel(excel_file) - -# 设置system的值 -system_value = "你是一个专业的,经验丰富的有心理学背景的职场教练。你总是根据有职场焦虑的病人的问题提供准确、全面和详细的答案。" - -# 将数据整理成jsonL格式 -json_data = [] -for index, row in df.iterrows(): - conversation = [ - { - "system": system_value, - "input": str(row['q']), - "output": str(row['a']) - } - ] - json_data.append({"conversation": conversation}) - -# 将json数据写入文件 -output_json_file = 'career_coach.jsonl' # 替换成实际的输出文件路径 -with open(output_json_file, 'w', encoding='utf-8') as f: - json.dump(json_data, f, ensure_ascii=False) - -print("JSONL文件生成成功!") - - -``` - -## 配置文件准备 - -Xtuner已经内置了许多的配置文件。可以通过Xtuner查看可配置文件 - -```bash -xtuner list-cfg -``` - -由于我们本次的基座微调模型为internLM2-chat-7b,所以我们可以查看Xtuner现在在InternLM2下已经支持了哪些配置文件 - -```bash -xtuner list-cfg |grep internlm2 -``` - -![4-2](images/4-2.png) - -```bash -# 复制配置文件 -xtuner copy-cfg internlm2_chat_7b_qlora_oasst1_e3 /root/autodl-tmp/ft-learn/config -# 修改配置文件名 -mv /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_copy.py /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_career_coach.py -``` - -复制完成之后要修改配置文件的几处参数 - -```bash -# PART 1 中 -# 预训练模型存放的位置 -pretrained_model_name_or_path = '/root/autodl-tmp/Shanghai_AI_Laboratory/internlm2-chat-7b' - -# 微调数据存放的位置 -data_path = '/root/autodl-tmp/ft-learn/dataset/career_coach.jsonl' - -# 训练中最大的文本长度 -max_length = 512 - -# 每一批训练样本的大小 -batch_size = 2 - -# 最大训练轮数 -max_epochs = 3 - -# 验证的频率 -evaluation_freq = 500 - -# 用于评估输出内容的问题(用于评估的问题尽量与数据集的question保持一致) -evaluation_inputs = [ -'我感到在职场中压力很大,总是焦虑不安,怎么办?', -'我在工作中总是害怕失败,怎样克服这种恐惧?', -'我感觉同事对我的期望很高,让我感到压力很大,怎么处理?' -] - - -# PART 3 中 -# 如果这里的如果没有修改的话,无法直接读取json文件 -dataset=dict(type=load_dataset, path='json', data_files=dict(train=data_path)) -# 这里也得改成None,否则会报错KeyError -dataset_map_fn=None - -``` - -## 模型微调 - -### 微调启动 - -```bash -xtuner train /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_career_coach.py --deepspeed deepspeed_zero2 -``` - -![4-3](images/4-3.png) - -训练完成之后,参数模型存放在`/root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/`目录下 - -### 模型转换成HF - -``` -# 新建模型存放的文件夹 -mkdir -p /root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/hf -# 添加环境变量 -export MKL_SERVICE_FORCE_INTEL=1 -# 模型转换 -xtuner convert pth_to_hf /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_career_coach.py /root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/iter_51.pth/ /root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/hf -``` - -### 合并HF adapter 到LLM - -```bash -mkdir -p /root/autodl-tmp/ft-learn/merged - -export MKL_SERVICE_FORCE_INTEL=1 -export MKL_THREADING_LAYER='GNU' - -# 原始模型参数存放的位置 -export NAME_OR_PATH_TO_LLM=/root/autodl-tmp/Shanghai_AI_Laboratory/internlm2-chat-7b - -# Hugging Face格式参数存放的位置 -export NAME_OR_PATH_TO_ADAPTER=/root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/hf - -# 最终Merge后的参数存放的位置 -mkdir -p /root/autodl-tmp/ft-learn/merged/internlm2_cc_hf_merge -export SAVE_PATH=/root/autodl-tmp/ft-learn/merged/internlm2_cc_hf_merge - - -# 执行参数Merge -xtuner convert merge \ - $NAME_OR_PATH_TO_LLM \ - $NAME_OR_PATH_TO_ADAPTER \ - $SAVE_PATH \ - --max-shard-size 2GB -``` - -![4-4](images/4-4.png) - -## Xtuner多轮对话介绍 - -XTuner 训练多轮对话模型时,采取了一种更加充分高效的方法,如下图所示。 - -
-Image -
- -我们将多轮对话进行拼接,之后输入模型,并行计算每个位置的 loss,而只有 Output 部分的 loss 参与回传。 - -XTuner 中多轮对话数据集格式如下所示: - -```json -[{ - "conversation":[ - { - "system": "You are an AI asssistant." - "input": "Hello?", - "output": "Hello! How can I help you?" - }, - { - "input": "What's the date today?", - "output": "Today is Monday, August 14, 2023." - }, - { - "input": "Thank you!", - "output": "You are welcome." - } - ] -}, -{ - "conversation":[ - { - "system": "You are an AI asssistant." - "input": "Hello?", - "output": "Hello! How can I help you?" - }, - { - "input": "How's the weather today in Rosso?", - "output": "The weather in Rosso on Wednesday, August 16th, is going to be cloudy for most of the day, together with moderate rain around noon." - }, - { - "input": "Thank you!", - "output": "You are welcome." - } - ] -}] -``` - -数据集中的 "conversation" 键对应的值是一个列表,用于保存每一轮对话的指令和实际回答(GroundTruth)。为了保持格式统一,增量预训练数据集和单轮对话数据集中的 "conversation" 键也对应一个列表,只不过该列表的长度为 1。而在多轮对话数据集中,"conversation" 列表的长度为 n,以容纳 n 轮的对话内容。 - -对多轮对话微调感兴趣的同学,也可以按照上面的数据格式进行数据微调。 - -## 写在最后 - +# InternLM2-7B-chat Xtuner Qlora 微调 + +## Xtuner介绍 +
+ +

+
+ +XTuner是上海人工智能实验室开发的低成本大模型训练工具箱,XTuner 是一个高效、灵活、全能的轻量化大模型微调工具库。只要**8G**。最低只需 **8GB** 显存,就可以微调InternLM2-7B模型,打造专属于你的 AI 助手。 + +仓库地址:https://github.com/InternLM/xtuner + +### Xtuner特点 + +**高效** + +- 支持大语言模型 LLM、多模态图文模型 VLM 的预训练及轻量级微调。XTuner 支持在 8GB 显存下微调 7B 模型,同时也支持多节点跨设备微调更大尺度模型(70B+)。 +- 自动分发高性能算子(如 FlashAttention、Triton kernels 等)以加速训练吞吐。 +- 兼容 [DeepSpeed](https://github.com/microsoft/DeepSpeed) 🚀,轻松应用各种 ZeRO 训练优化策略。 + +**灵活** + +- 支持多种大语言模型,包括但不限于 [InternLM](https://huggingface.co/internlm)、[Mixtral-8x7B](https://huggingface.co/mistralai)、[Llama2](https://huggingface.co/meta-llama)、[ChatGLM](https://huggingface.co/THUDM)、[Qwen](https://huggingface.co/Qwen)、[Baichuan](https://huggingface.co/baichuan-inc)。 +- 支持多模态图文模型 LLaVA 的预训练与微调。利用 XTuner 训得模型 [LLaVA-InternLM2-20B](https://huggingface.co/xtuner/llava-internlm2-20b) 表现优异。 +- 精心设计的数据管道,兼容任意数据格式,开源数据或自定义数据皆可快速上手。 +- 支持 [QLoRA](http://arxiv.org/abs/2305.14314)、[LoRA](http://arxiv.org/abs/2106.09685)、全量参数微调等多种微调算法,支撑用户根据具体需求作出最优选择。 + +**全能** + +- 支持增量预训练、指令微调与 Agent 微调。 +- 预定义众多开源对话模版,支持与开源或训练所得模型进行对话。 +- 训练所得模型可无缝接入部署工具库 [LMDeploy](https://github.com/InternLM/lmdeploy)、大规模评测工具库 [OpenCompass](https://github.com/open-compass/opencompass) 及 [VLMEvalKit](https://github.com/open-compass/VLMEvalKit)。 + +## 环境准备 + +在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 。 + +接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 + +![机器配置选择](./images/1.png) + +### 创建工作目录 + +创建本次微调实践的工作目录`/root/autodl-tmp/ft-learn` + +``` +# 创建微调工作目录 +mkdir -p /root/autodl-tmp/ft-learn + +# 创建微调数据集存放目录 +mkdir -p /root/autodl-tmp/ft-learn/dataset + +# 创建微调配置文件存放目录 +mkdir -p /root/autodl-tmp/ft-learn/config + +``` + +### 安装依赖 + +```bash +# 升级pip +python -m pip install --upgrade pip +# 安装python依赖 +pip install modelscope==1.9.5 +pip install transformers==4.36.2 +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.24.1 +pip install transformers_stream_generator==0.0.4 +pip install einops ujson +pip install protobuf +``` + +### 使用modelscope下载模型 + +使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 + +在`Internlm2-7b-chat-web.ipynb`文件中新建一个代码块,运行下载`internlm2-chat-7b`模型。模型下载需要时间,我们直接往下看 + +``` +from modelscope import snapshot_download + +model_dir = snapshot_download('Shanghai_AI_Laboratory/internlm2-chat-7b', cache_dir='/root/autodl-tmp', revision='master') +``` + +![3-3](images/3-3.png) + +### 安装Xtuner + +```bash +git clone -b v0.1.14 https://github.com/InternLM/xtuner +cd xtuner +# 从源码安装 XTuner +pip install -e '.[all]' +# 安装完成之后就可以在命令行使用xtuner了 +# 查看xtuner使用帮助 +xtuner help +# 查看xtuner版本 +xtuner version +``` + +![4-1](images/4-1.png) + +## 数据集处理 + +我自己整理的`心理大模型-职场焦虑语料.xlsx`,通过`gen_qa_json.py`文件生成一个`career_coach.jsonl`文件 + +运行`python /root/autodl-tmp/ft-learn/dataset/gen_qa_json.py`生成文件,你们也可以按照我的数据语料格式,自定义你们自己的数据集。`gen_qa_json.py`文件代码如下: + +``` +import pandas as pd +import json + +# 读取Excel文件 +excel_file = './心理大模型-职场焦虑语料.xlsx' # 替换成实际的Excel文件路径 +df = pd.read_excel(excel_file) + +# 设置system的值 +system_value = "你是一个专业的,经验丰富的有心理学背景的职场教练。你总是根据有职场焦虑的病人的问题提供准确、全面和详细的答案。" + +# 将数据整理成jsonL格式 +json_data = [] +for index, row in df.iterrows(): + conversation = [ + { + "system": system_value, + "input": str(row['q']), + "output": str(row['a']) + } + ] + json_data.append({"conversation": conversation}) + +# 将json数据写入文件 +output_json_file = 'career_coach.jsonl' # 替换成实际的输出文件路径 +with open(output_json_file, 'w', encoding='utf-8') as f: + json.dump(json_data, f, ensure_ascii=False) + +print("JSONL文件生成成功!") + + +``` + +## 配置文件准备 + +Xtuner已经内置了许多的配置文件。可以通过Xtuner查看可配置文件 + +```bash +xtuner list-cfg +``` + +由于我们本次的基座微调模型为internLM2-chat-7b,所以我们可以查看Xtuner现在在InternLM2下已经支持了哪些配置文件 + +```bash +xtuner list-cfg |grep internlm2 +``` + +![4-2](images/4-2.png) + +```bash +# 复制配置文件 +xtuner copy-cfg internlm2_chat_7b_qlora_oasst1_e3 /root/autodl-tmp/ft-learn/config +# 修改配置文件名 +mv /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_copy.py /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_career_coach.py +``` + +复制完成之后要修改配置文件的几处参数 + +```bash +# PART 1 中 +# 预训练模型存放的位置 +pretrained_model_name_or_path = '/root/autodl-tmp/Shanghai_AI_Laboratory/internlm2-chat-7b' + +# 微调数据存放的位置 +data_path = '/root/autodl-tmp/ft-learn/dataset/career_coach.jsonl' + +# 训练中最大的文本长度 +max_length = 512 + +# 每一批训练样本的大小 +batch_size = 2 + +# 最大训练轮数 +max_epochs = 3 + +# 验证的频率 +evaluation_freq = 500 + +# 用于评估输出内容的问题(用于评估的问题尽量与数据集的question保持一致) +evaluation_inputs = [ +'我感到在职场中压力很大,总是焦虑不安,怎么办?', +'我在工作中总是害怕失败,怎样克服这种恐惧?', +'我感觉同事对我的期望很高,让我感到压力很大,怎么处理?' +] + + +# PART 3 中 +# 如果这里的如果没有修改的话,无法直接读取json文件 +dataset=dict(type=load_dataset, path='json', data_files=dict(train=data_path)) +# 这里也得改成None,否则会报错KeyError +dataset_map_fn=None + +``` + +## 模型微调 + +### 微调启动 + +```bash +xtuner train /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_career_coach.py --deepspeed deepspeed_zero2 +``` + +![4-3](images/4-3.png) + +训练完成之后,参数模型存放在`/root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/`目录下 + +### 模型转换成HF + +``` +# 新建模型存放的文件夹 +mkdir -p /root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/hf +# 添加环境变量 +export MKL_SERVICE_FORCE_INTEL=1 +# 模型转换 +xtuner convert pth_to_hf /root/autodl-tmp/ft-learn/config/internlm2_chat_7b_qlora_oasst1_e3_career_coach.py /root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/iter_51.pth/ /root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/hf +``` + +### 合并HF adapter 到LLM + +```bash +mkdir -p /root/autodl-tmp/ft-learn/merged + +export MKL_SERVICE_FORCE_INTEL=1 +export MKL_THREADING_LAYER='GNU' + +# 原始模型参数存放的位置 +export NAME_OR_PATH_TO_LLM=/root/autodl-tmp/Shanghai_AI_Laboratory/internlm2-chat-7b + +# Hugging Face格式参数存放的位置 +export NAME_OR_PATH_TO_ADAPTER=/root/work_dirs/internlm2_chat_7b_qlora_oasst1_e3_career_coach/hf + +# 最终Merge后的参数存放的位置 +mkdir -p /root/autodl-tmp/ft-learn/merged/internlm2_cc_hf_merge +export SAVE_PATH=/root/autodl-tmp/ft-learn/merged/internlm2_cc_hf_merge + + +# 执行参数Merge +xtuner convert merge \ + $NAME_OR_PATH_TO_LLM \ + $NAME_OR_PATH_TO_ADAPTER \ + $SAVE_PATH \ + --max-shard-size 2GB +``` + +![4-4](images/4-4.png) + +## Xtuner多轮对话介绍 + +XTuner 训练多轮对话模型时,采取了一种更加充分高效的方法,如下图所示。 + +
+Image +
+ +我们将多轮对话进行拼接,之后输入模型,并行计算每个位置的 loss,而只有 Output 部分的 loss 参与回传。 + +XTuner 中多轮对话数据集格式如下所示: + +```json +[{ + "conversation":[ + { + "system": "You are an AI asssistant." + "input": "Hello?", + "output": "Hello! How can I help you?" + }, + { + "input": "What's the date today?", + "output": "Today is Monday, August 14, 2023." + }, + { + "input": "Thank you!", + "output": "You are welcome." + } + ] +}, +{ + "conversation":[ + { + "system": "You are an AI asssistant." + "input": "Hello?", + "output": "Hello! How can I help you?" + }, + { + "input": "How's the weather today in Rosso?", + "output": "The weather in Rosso on Wednesday, August 16th, is going to be cloudy for most of the day, together with moderate rain around noon." + }, + { + "input": "Thank you!", + "output": "You are welcome." + } + ] +}] +``` + +数据集中的 "conversation" 键对应的值是一个列表,用于保存每一轮对话的指令和实际回答(GroundTruth)。为了保持格式统一,增量预训练数据集和单轮对话数据集中的 "conversation" 键也对应一个列表,只不过该列表的长度为 1。而在多轮对话数据集中,"conversation" 列表的长度为 n,以容纳 n 轮的对话内容。 + +对多轮对话微调感兴趣的同学,也可以按照上面的数据格式进行数据微调。 + +## 写在最后 + 本节关于Xtuner的微调步骤中提到的职场焦虑数据语料,是我用于开源项目职场教练大模型微调时所使用的语料,感兴趣的同学也可看一看我的这个开源项目[career_coach](https://github.com/BaiYu96/career_coach),欢迎点个star。项目的data部分有介绍到多轮对话数据的整理与生成,其实与本教程是一样的。 \ No newline at end of file diff --git a/InternLM2/dataset/心理大模型-职场焦虑语料.xlsx b/models/InternLM2/dataset/心理大模型-职场焦虑语料.xlsx similarity index 100% rename from InternLM2/dataset/心理大模型-职场焦虑语料.xlsx rename to models/InternLM2/dataset/心理大模型-职场焦虑语料.xlsx diff --git a/InternLM2/images/1.png b/models/InternLM2/images/1.png similarity index 100% rename from InternLM2/images/1.png rename to models/InternLM2/images/1.png diff --git a/InternLM2/images/2.png b/models/InternLM2/images/2.png similarity index 100% rename from InternLM2/images/2.png rename to models/InternLM2/images/2.png diff --git a/InternLM2/images/3-1.png b/models/InternLM2/images/3-1.png similarity index 100% rename from InternLM2/images/3-1.png rename to models/InternLM2/images/3-1.png diff --git a/InternLM2/images/3-2.png b/models/InternLM2/images/3-2.png similarity index 100% rename from InternLM2/images/3-2.png rename to models/InternLM2/images/3-2.png diff --git a/InternLM2/images/3-3.png b/models/InternLM2/images/3-3.png similarity index 100% rename from InternLM2/images/3-3.png rename to models/InternLM2/images/3-3.png diff --git a/InternLM2/images/3-4.png b/models/InternLM2/images/3-4.png similarity index 100% rename from InternLM2/images/3-4.png rename to models/InternLM2/images/3-4.png diff --git a/InternLM2/images/3-5.png b/models/InternLM2/images/3-5.png similarity index 100% rename from InternLM2/images/3-5.png rename to models/InternLM2/images/3-5.png diff --git a/InternLM2/images/3-6.png b/models/InternLM2/images/3-6.png similarity index 100% rename from InternLM2/images/3-6.png rename to models/InternLM2/images/3-6.png diff --git a/InternLM2/images/3-7.png b/models/InternLM2/images/3-7.png similarity index 100% rename from InternLM2/images/3-7.png rename to models/InternLM2/images/3-7.png diff --git a/InternLM2/images/3-8.png b/models/InternLM2/images/3-8.png similarity index 100% rename from InternLM2/images/3-8.png rename to models/InternLM2/images/3-8.png diff --git a/InternLM2/images/3.png b/models/InternLM2/images/3.png similarity index 100% rename from InternLM2/images/3.png rename to models/InternLM2/images/3.png diff --git a/InternLM2/images/4-1.png b/models/InternLM2/images/4-1.png similarity index 100% rename from InternLM2/images/4-1.png rename to models/InternLM2/images/4-1.png diff --git a/InternLM2/images/4-2.png b/models/InternLM2/images/4-2.png similarity index 100% rename from InternLM2/images/4-2.png rename to models/InternLM2/images/4-2.png diff --git a/InternLM2/images/4-3.png b/models/InternLM2/images/4-3.png similarity index 100% rename from InternLM2/images/4-3.png rename to models/InternLM2/images/4-3.png diff --git a/InternLM2/images/4-4.png b/models/InternLM2/images/4-4.png similarity index 100% rename from InternLM2/images/4-4.png rename to models/InternLM2/images/4-4.png diff --git a/LLaMA3/01-LLaMA3-8B-Instruct FastApi 部署调用.md b/models/LLaMA3/01-LLaMA3-8B-Instruct FastApi 部署调用.md similarity index 100% rename from LLaMA3/01-LLaMA3-8B-Instruct FastApi 部署调用.md rename to models/LLaMA3/01-LLaMA3-8B-Instruct FastApi 部署调用.md diff --git a/LLaMA3/02-LLaMA3-8B-Instruct langchain 接入.md b/models/LLaMA3/02-LLaMA3-8B-Instruct langchain 接入.md similarity index 100% rename from LLaMA3/02-LLaMA3-8B-Instruct langchain 接入.md rename to models/LLaMA3/02-LLaMA3-8B-Instruct langchain 接入.md diff --git a/LLaMA3/03-LLaMA3-8B-Instruct WebDemo 部署.md b/models/LLaMA3/03-LLaMA3-8B-Instruct WebDemo 部署.md similarity index 100% rename from LLaMA3/03-LLaMA3-8B-Instruct WebDemo 部署.md rename to models/LLaMA3/03-LLaMA3-8B-Instruct WebDemo 部署.md diff --git a/LLaMA3/04-LLaMA3-8B-Instruct Lora 微调.md b/models/LLaMA3/04-LLaMA3-8B-Instruct Lora 微调.md similarity index 100% rename from LLaMA3/04-LLaMA3-8B-Instruct Lora 微调.md rename to models/LLaMA3/04-LLaMA3-8B-Instruct Lora 微调.md diff --git a/LLaMA3/LLaMA3-8B-Instruct Lora.ipynb b/models/LLaMA3/LLaMA3-8B-Instruct Lora.ipynb similarity index 100% rename from LLaMA3/LLaMA3-8B-Instruct Lora.ipynb rename to models/LLaMA3/LLaMA3-8B-Instruct Lora.ipynb diff --git a/LLaMA3/images/api_resp.png b/models/LLaMA3/images/api_resp.png similarity index 100% rename from LLaMA3/images/api_resp.png rename to models/LLaMA3/images/api_resp.png diff --git a/LLaMA3/images/api_start.png b/models/LLaMA3/images/api_start.png similarity index 100% rename from LLaMA3/images/api_start.png rename to models/LLaMA3/images/api_start.png diff --git a/LLaMA3/images/image-1.png b/models/LLaMA3/images/image-1.png similarity index 100% rename from LLaMA3/images/image-1.png rename to models/LLaMA3/images/image-1.png diff --git a/LLaMA3/images/image-2.png b/models/LLaMA3/images/image-2.png similarity index 100% rename from LLaMA3/images/image-2.png rename to models/LLaMA3/images/image-2.png diff --git a/LLaMA3/images/image-3.png b/models/LLaMA3/images/image-3.png similarity index 100% rename from LLaMA3/images/image-3.png rename to models/LLaMA3/images/image-3.png diff --git a/MiniCPM/MiniCPM-2B-chat FastApi 部署调用.md b/models/MiniCPM/MiniCPM-2B-chat FastApi 部署调用.md similarity index 100% rename from MiniCPM/MiniCPM-2B-chat FastApi 部署调用.md rename to models/MiniCPM/MiniCPM-2B-chat FastApi 部署调用.md diff --git a/MiniCPM/MiniCPM-2B-chat Lora && Full 微调.md b/models/MiniCPM/MiniCPM-2B-chat Lora && Full 微调.md similarity index 100% rename from MiniCPM/MiniCPM-2B-chat Lora && Full 微调.md rename to models/MiniCPM/MiniCPM-2B-chat Lora && Full 微调.md diff --git a/MiniCPM/MiniCPM-2B-chat WebDemo部署.md b/models/MiniCPM/MiniCPM-2B-chat WebDemo部署.md similarity index 100% rename from MiniCPM/MiniCPM-2B-chat WebDemo部署.md rename to models/MiniCPM/MiniCPM-2B-chat WebDemo部署.md diff --git a/MiniCPM/MiniCPM-2B-chat langchain接入.md b/models/MiniCPM/MiniCPM-2B-chat langchain接入.md similarity index 100% rename from MiniCPM/MiniCPM-2B-chat langchain接入.md rename to models/MiniCPM/MiniCPM-2B-chat langchain接入.md diff --git a/MiniCPM/MiniCPM-2B-chat transformers 部署调用.md b/models/MiniCPM/MiniCPM-2B-chat transformers 部署调用.md similarity index 100% rename from MiniCPM/MiniCPM-2B-chat transformers 部署调用.md rename to models/MiniCPM/MiniCPM-2B-chat transformers 部署调用.md diff --git a/MiniCPM/ds_config.json b/models/MiniCPM/ds_config.json similarity index 100% rename from MiniCPM/ds_config.json rename to models/MiniCPM/ds_config.json diff --git a/MiniCPM/images/image-1.png b/models/MiniCPM/images/image-1.png similarity index 100% rename from MiniCPM/images/image-1.png rename to models/MiniCPM/images/image-1.png diff --git a/MiniCPM/images/image-10.png b/models/MiniCPM/images/image-10.png similarity index 100% rename from MiniCPM/images/image-10.png rename to models/MiniCPM/images/image-10.png diff --git a/MiniCPM/images/image-2.png b/models/MiniCPM/images/image-2.png similarity index 100% rename from MiniCPM/images/image-2.png rename to models/MiniCPM/images/image-2.png diff --git a/MiniCPM/images/image-3.png b/models/MiniCPM/images/image-3.png similarity index 100% rename from MiniCPM/images/image-3.png rename to models/MiniCPM/images/image-3.png diff --git a/MiniCPM/images/image-4.png b/models/MiniCPM/images/image-4.png similarity index 100% rename from MiniCPM/images/image-4.png rename to models/MiniCPM/images/image-4.png diff --git a/MiniCPM/images/image-5.png b/models/MiniCPM/images/image-5.png similarity index 100% rename from MiniCPM/images/image-5.png rename to models/MiniCPM/images/image-5.png diff --git a/MiniCPM/images/image-6.png b/models/MiniCPM/images/image-6.png similarity index 100% rename from MiniCPM/images/image-6.png rename to models/MiniCPM/images/image-6.png diff --git a/MiniCPM/images/image-7.png b/models/MiniCPM/images/image-7.png similarity index 100% rename from MiniCPM/images/image-7.png rename to models/MiniCPM/images/image-7.png diff --git a/MiniCPM/images/image-8.png b/models/MiniCPM/images/image-8.png similarity index 100% rename from MiniCPM/images/image-8.png rename to models/MiniCPM/images/image-8.png diff --git a/MiniCPM/images/image-9.png b/models/MiniCPM/images/image-9.png similarity index 100% rename from MiniCPM/images/image-9.png rename to models/MiniCPM/images/image-9.png diff --git a/MiniCPM/train.py b/models/MiniCPM/train.py similarity index 100% rename from MiniCPM/train.py rename to models/MiniCPM/train.py diff --git a/MiniCPM/train.sh b/models/MiniCPM/train.sh similarity index 100% rename from MiniCPM/train.sh rename to models/MiniCPM/train.sh diff --git a/Qwen-Audio/01-Qwen-Audio-chat FastApi.md b/models/Qwen-Audio/01-Qwen-Audio-chat FastApi.md similarity index 100% rename from Qwen-Audio/01-Qwen-Audio-chat FastApi.md rename to models/Qwen-Audio/01-Qwen-Audio-chat FastApi.md diff --git a/Qwen-Audio/02-Qwen-Audio-chat WebDemo.md b/models/Qwen-Audio/02-Qwen-Audio-chat WebDemo.md similarity index 100% rename from Qwen-Audio/02-Qwen-Audio-chat WebDemo.md rename to models/Qwen-Audio/02-Qwen-Audio-chat WebDemo.md diff --git a/Qwen-Audio/images/image-1.png b/models/Qwen-Audio/images/image-1.png similarity index 100% rename from Qwen-Audio/images/image-1.png rename to models/Qwen-Audio/images/image-1.png diff --git a/Qwen-Audio/images/image-2.png b/models/Qwen-Audio/images/image-2.png similarity index 100% rename from Qwen-Audio/images/image-2.png rename to models/Qwen-Audio/images/image-2.png diff --git a/Qwen-Audio/images/image-3.png b/models/Qwen-Audio/images/image-3.png similarity index 100% rename from Qwen-Audio/images/image-3.png rename to models/Qwen-Audio/images/image-3.png diff --git a/Qwen-Audio/images/image-4.png b/models/Qwen-Audio/images/image-4.png similarity index 100% rename from Qwen-Audio/images/image-4.png rename to models/Qwen-Audio/images/image-4.png diff --git a/Qwen/01-Qwen-7B-Chat Transformers部署调用.md b/models/Qwen/01-Qwen-7B-Chat Transformers部署调用.md similarity index 100% rename from Qwen/01-Qwen-7B-Chat Transformers部署调用.md rename to models/Qwen/01-Qwen-7B-Chat Transformers部署调用.md diff --git a/Qwen/02-Qwen-7B-Chat FastApi 部署调用.md b/models/Qwen/02-Qwen-7B-Chat FastApi 部署调用.md similarity index 100% rename from Qwen/02-Qwen-7B-Chat FastApi 部署调用.md rename to models/Qwen/02-Qwen-7B-Chat FastApi 部署调用.md diff --git a/Qwen/03-Qwen-7B-Chat WebDemo.md b/models/Qwen/03-Qwen-7B-Chat WebDemo.md similarity index 100% rename from Qwen/03-Qwen-7B-Chat WebDemo.md rename to models/Qwen/03-Qwen-7B-Chat WebDemo.md diff --git a/Qwen/04-Qwen-7B-Chat Lora 微调.ipynb b/models/Qwen/04-Qwen-7B-Chat Lora 微调.ipynb similarity index 100% rename from Qwen/04-Qwen-7B-Chat Lora 微调.ipynb rename to models/Qwen/04-Qwen-7B-Chat Lora 微调.ipynb diff --git a/Qwen/04-Qwen-7B-Chat Lora 微调.md b/models/Qwen/04-Qwen-7B-Chat Lora 微调.md similarity index 100% rename from Qwen/04-Qwen-7B-Chat Lora 微调.md rename to models/Qwen/04-Qwen-7B-Chat Lora 微调.md diff --git a/Qwen/04-Qwen-7B-Chat Lora 微调.py b/models/Qwen/04-Qwen-7B-Chat Lora 微调.py similarity index 100% rename from Qwen/04-Qwen-7B-Chat Lora 微调.py rename to models/Qwen/04-Qwen-7B-Chat Lora 微调.py diff --git a/Qwen/05-Qwen-7B-Chat Ptuning 微调.md b/models/Qwen/05-Qwen-7B-Chat Ptuning 微调.md similarity index 100% rename from Qwen/05-Qwen-7B-Chat Ptuning 微调.md rename to models/Qwen/05-Qwen-7B-Chat Ptuning 微调.md diff --git a/Qwen/05-Qwen-7B-Chat Ptuning 微调.py b/models/Qwen/05-Qwen-7B-Chat Ptuning 微调.py similarity index 100% rename from Qwen/05-Qwen-7B-Chat Ptuning 微调.py rename to models/Qwen/05-Qwen-7B-Chat Ptuning 微调.py diff --git a/Qwen/06-Qwen-7B-chat 全量微调.md b/models/Qwen/06-Qwen-7B-chat 全量微调.md similarity index 100% rename from Qwen/06-Qwen-7B-chat 全量微调.md rename to models/Qwen/06-Qwen-7B-chat 全量微调.md diff --git a/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手.md b/models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手.md similarity index 100% rename from Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手.md rename to models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手.md diff --git a/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/LLM.py b/models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/LLM.py similarity index 100% rename from Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/LLM.py rename to models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/LLM.py diff --git a/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/creat_db.py b/models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/creat_db.py similarity index 100% rename from Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/creat_db.py rename to models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/creat_db.py diff --git a/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/readme.md b/models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/readme.md similarity index 100% rename from Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/readme.md rename to models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/readme.md diff --git a/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/run_gradio.py b/models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/run_gradio.py similarity index 100% rename from Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/run_gradio.py rename to models/Qwen/07-Qwen-7B-Chat 接入langchain搭建知识库助手/run_gradio.py diff --git a/Qwen/08-Qwen-7B-Chat Lora -4bit微调.ipynb b/models/Qwen/08-Qwen-7B-Chat Lora -4bit微调.ipynb similarity index 100% rename from Qwen/08-Qwen-7B-Chat Lora -4bit微调.ipynb rename to models/Qwen/08-Qwen-7B-Chat Lora -4bit微调.ipynb diff --git a/Qwen/08-Qwen-7B-Chat Lora -8bit微调.ipynb b/models/Qwen/08-Qwen-7B-Chat Lora -8bit微调.ipynb similarity index 100% rename from Qwen/08-Qwen-7B-Chat Lora -8bit微调.ipynb rename to models/Qwen/08-Qwen-7B-Chat Lora -8bit微调.ipynb diff --git a/Qwen/08-Qwen-7B-Chat Lora 低精度微调.md b/models/Qwen/08-Qwen-7B-Chat Lora 低精度微调.md similarity index 100% rename from Qwen/08-Qwen-7B-Chat Lora 低精度微调.md rename to models/Qwen/08-Qwen-7B-Chat Lora 低精度微调.md diff --git a/Qwen/08-Qwen-7B-Chat Lora 低精度微调.py b/models/Qwen/08-Qwen-7B-Chat Lora 低精度微调.py similarity index 100% rename from Qwen/08-Qwen-7B-Chat Lora 低精度微调.py rename to models/Qwen/08-Qwen-7B-Chat Lora 低精度微调.py diff --git a/Qwen/09-Qwen-1_8B-chat CPU 部署 .ipynb b/models/Qwen/09-Qwen-1_8B-chat CPU 部署 .ipynb similarity index 100% rename from Qwen/09-Qwen-1_8B-chat CPU 部署 .ipynb rename to models/Qwen/09-Qwen-1_8B-chat CPU 部署 .ipynb diff --git a/Qwen/09-Qwen-1_8B-chat CPU 部署 .md b/models/Qwen/09-Qwen-1_8B-chat CPU 部署 .md similarity index 100% rename from Qwen/09-Qwen-1_8B-chat CPU 部署 .md rename to models/Qwen/09-Qwen-1_8B-chat CPU 部署 .md diff --git a/Qwen/environment.yml b/models/Qwen/environment.yml similarity index 100% rename from Qwen/environment.yml rename to models/Qwen/environment.yml diff --git a/Qwen/images/1.png b/models/Qwen/images/1.png similarity index 100% rename from Qwen/images/1.png rename to models/Qwen/images/1.png diff --git a/Qwen/images/2.png b/models/Qwen/images/2.png similarity index 100% rename from Qwen/images/2.png rename to models/Qwen/images/2.png diff --git a/Qwen/images/3.png b/models/Qwen/images/3.png similarity index 100% rename from Qwen/images/3.png rename to models/Qwen/images/3.png diff --git a/Qwen/images/4.png b/models/Qwen/images/4.png similarity index 100% rename from Qwen/images/4.png rename to models/Qwen/images/4.png diff --git a/Qwen/images/5.png b/models/Qwen/images/5.png similarity index 100% rename from Qwen/images/5.png rename to models/Qwen/images/5.png diff --git a/Qwen/images/6.png b/models/Qwen/images/6.png similarity index 100% rename from Qwen/images/6.png rename to models/Qwen/images/6.png diff --git a/Qwen/images/7.png b/models/Qwen/images/7.png similarity index 100% rename from Qwen/images/7.png rename to models/Qwen/images/7.png diff --git a/Qwen/images/8.png b/models/Qwen/images/8.png similarity index 100% rename from Qwen/images/8.png rename to models/Qwen/images/8.png diff --git a/Qwen/images/P-tuning.png b/models/Qwen/images/P-tuning.png similarity index 100% rename from Qwen/images/P-tuning.png rename to models/Qwen/images/P-tuning.png diff --git a/Qwen1.5/01-Qwen1.5-7B-Chat FastApi 部署调用.md b/models/Qwen1.5/01-Qwen1.5-7B-Chat FastApi 部署调用.md similarity index 100% rename from Qwen1.5/01-Qwen1.5-7B-Chat FastApi 部署调用.md rename to models/Qwen1.5/01-Qwen1.5-7B-Chat FastApi 部署调用.md diff --git a/Qwen1.5/02-Qwen1.5-7B-Chat 接入langchain搭建知识库助手.md b/models/Qwen1.5/02-Qwen1.5-7B-Chat 接入langchain搭建知识库助手.md similarity index 100% rename from Qwen1.5/02-Qwen1.5-7B-Chat 接入langchain搭建知识库助手.md rename to models/Qwen1.5/02-Qwen1.5-7B-Chat 接入langchain搭建知识库助手.md diff --git a/Qwen1.5/03-Qwen1.5-7B-Chat WebDemo.md b/models/Qwen1.5/03-Qwen1.5-7B-Chat WebDemo.md similarity index 100% rename from Qwen1.5/03-Qwen1.5-7B-Chat WebDemo.md rename to models/Qwen1.5/03-Qwen1.5-7B-Chat WebDemo.md diff --git a/Qwen1.5/04-Qwen1.5-7B-chat Lora 微调.md b/models/Qwen1.5/04-Qwen1.5-7B-chat Lora 微调.md similarity index 100% rename from Qwen1.5/04-Qwen1.5-7B-chat Lora 微调.md rename to models/Qwen1.5/04-Qwen1.5-7B-chat Lora 微调.md diff --git a/Qwen1.5/05-Qwen1.5-7B-Chat-GPTQ-Int4 WebDemo.md b/models/Qwen1.5/05-Qwen1.5-7B-Chat-GPTQ-Int4 WebDemo.md similarity index 100% rename from Qwen1.5/05-Qwen1.5-7B-Chat-GPTQ-Int4 WebDemo.md rename to models/Qwen1.5/05-Qwen1.5-7B-Chat-GPTQ-Int4 WebDemo.md diff --git a/Qwen1.5/06-Qwen1.5-MoE-A2.7B.md b/models/Qwen1.5/06-Qwen1.5-MoE-A2.7B.md similarity index 100% rename from Qwen1.5/06-Qwen1.5-MoE-A2.7B.md rename to models/Qwen1.5/06-Qwen1.5-MoE-A2.7B.md diff --git a/Qwen1.5/07-Qwen1.5-7B-Chat vLLM 推理部署调用.md b/models/Qwen1.5/07-Qwen1.5-7B-Chat vLLM 推理部署调用.md similarity index 100% rename from Qwen1.5/07-Qwen1.5-7B-Chat vLLM 推理部署调用.md rename to models/Qwen1.5/07-Qwen1.5-7B-Chat vLLM 推理部署调用.md diff --git a/Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.ipynb b/models/Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.ipynb similarity index 100% rename from Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.ipynb rename to models/Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.ipynb diff --git a/Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.md b/models/Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.md similarity index 100% rename from Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.md rename to models/Qwen1.5/08-Qwen1.5-7B-chat LoRA微调接入实验管理.md diff --git a/Qwen1.5/Qwen1.5-7B-Chat Lora.ipynb b/models/Qwen1.5/Qwen1.5-7B-Chat Lora.ipynb similarity index 100% rename from Qwen1.5/Qwen1.5-7B-Chat Lora.ipynb rename to models/Qwen1.5/Qwen1.5-7B-Chat Lora.ipynb diff --git a/Qwen1.5/benchmark_throughput.py b/models/Qwen1.5/benchmark_throughput.py similarity index 100% rename from Qwen1.5/benchmark_throughput.py rename to models/Qwen1.5/benchmark_throughput.py diff --git a/Qwen1.5/images/2.png b/models/Qwen1.5/images/2.png similarity index 100% rename from Qwen1.5/images/2.png rename to models/Qwen1.5/images/2.png diff --git a/Qwen1.5/images/6.png b/models/Qwen1.5/images/6.png similarity index 100% rename from Qwen1.5/images/6.png rename to models/Qwen1.5/images/6.png diff --git a/Qwen1.5/images/Qwen1.5-7b-gptq-int4-1.png b/models/Qwen1.5/images/Qwen1.5-7b-gptq-int4-1.png similarity index 100% rename from Qwen1.5/images/Qwen1.5-7b-gptq-int4-1.png rename to models/Qwen1.5/images/Qwen1.5-7b-gptq-int4-1.png diff --git a/Qwen1.5/images/Qwen1.5-7b-gptq-int4-2.png b/models/Qwen1.5/images/Qwen1.5-7b-gptq-int4-2.png similarity index 100% rename from Qwen1.5/images/Qwen1.5-7b-gptq-int4-2.png rename to models/Qwen1.5/images/Qwen1.5-7b-gptq-int4-2.png diff --git a/Qwen1.5/images/Qwen1.5-vllm-api-stat.png b/models/Qwen1.5/images/Qwen1.5-vllm-api-stat.png similarity index 100% rename from Qwen1.5/images/Qwen1.5-vllm-api-stat.png rename to models/Qwen1.5/images/Qwen1.5-vllm-api-stat.png diff --git a/Qwen1.5/images/Qwen1.5-vllm-gpu-select.png b/models/Qwen1.5/images/Qwen1.5-vllm-gpu-select.png similarity index 100% rename from Qwen1.5/images/Qwen1.5-vllm-gpu-select.png rename to models/Qwen1.5/images/Qwen1.5-vllm-gpu-select.png diff --git a/Qwen1.5/images/Qwen1.5-vllm.png b/models/Qwen1.5/images/Qwen1.5-vllm.png similarity index 100% rename from Qwen1.5/images/Qwen1.5-vllm.png rename to models/Qwen1.5/images/Qwen1.5-vllm.png diff --git a/Qwen1.5/images/Qwen2-Web1.png b/models/Qwen1.5/images/Qwen2-Web1.png similarity index 100% rename from Qwen1.5/images/Qwen2-Web1.png rename to models/Qwen1.5/images/Qwen2-Web1.png diff --git a/Qwen1.5/images/Qwen2-Web2.png b/models/Qwen1.5/images/Qwen2-Web2.png similarity index 100% rename from Qwen1.5/images/Qwen2-Web2.png rename to models/Qwen1.5/images/Qwen2-Web2.png diff --git a/Qwen1.5/images/image-2.png b/models/Qwen1.5/images/image-2.png similarity index 100% rename from Qwen1.5/images/image-2.png rename to models/Qwen1.5/images/image-2.png diff --git a/Qwen1.5/images/question_to_the_Qwen2.png b/models/Qwen1.5/images/question_to_the_Qwen2.png similarity index 100% rename from Qwen1.5/images/question_to_the_Qwen2.png rename to models/Qwen1.5/images/question_to_the_Qwen2.png diff --git a/Qwen1.5/images/swanlabcallbacks.png b/models/Qwen1.5/images/swanlabcallbacks.png similarity index 100% rename from Qwen1.5/images/swanlabcallbacks.png rename to models/Qwen1.5/images/swanlabcallbacks.png diff --git a/Qwen1.5/images/swanlabchart.png b/models/Qwen1.5/images/swanlabchart.png similarity index 100% rename from Qwen1.5/images/swanlabchart.png rename to models/Qwen1.5/images/swanlabchart.png diff --git a/Qwen1.5/images/swanlabdisplay.png b/models/Qwen1.5/images/swanlabdisplay.png similarity index 100% rename from Qwen1.5/images/swanlabdisplay.png rename to models/Qwen1.5/images/swanlabdisplay.png diff --git a/Qwen1.5/images/swanlabsettings.png b/models/Qwen1.5/images/swanlabsettings.png similarity index 100% rename from Qwen1.5/images/swanlabsettings.png rename to models/Qwen1.5/images/swanlabsettings.png diff --git a/Qwen1.5/images/swanlabweb.png b/models/Qwen1.5/images/swanlabweb.png similarity index 100% rename from Qwen1.5/images/swanlabweb.png rename to models/Qwen1.5/images/swanlabweb.png diff --git a/Qwen2/01-Qwen2-7B-Instruct FastApi 部署调用.md b/models/Qwen2/01-Qwen2-7B-Instruct FastApi 部署调用.md similarity index 100% rename from Qwen2/01-Qwen2-7B-Instruct FastApi 部署调用.md rename to models/Qwen2/01-Qwen2-7B-Instruct FastApi 部署调用.md diff --git a/Qwen2/02-Qwen2-7B-Instruct Langchain 接入.md b/models/Qwen2/02-Qwen2-7B-Instruct Langchain 接入.md similarity index 100% rename from Qwen2/02-Qwen2-7B-Instruct Langchain 接入.md rename to models/Qwen2/02-Qwen2-7B-Instruct Langchain 接入.md diff --git a/Qwen2/03-Qwen2-7B-Instruct WebDemo部署.md b/models/Qwen2/03-Qwen2-7B-Instruct WebDemo部署.md similarity index 97% rename from Qwen2/03-Qwen2-7B-Instruct WebDemo部署.md rename to models/Qwen2/03-Qwen2-7B-Instruct WebDemo部署.md index 90308bf..54b77a0 100644 --- a/Qwen2/03-Qwen2-7B-Instruct WebDemo部署.md +++ b/models/Qwen2/03-Qwen2-7B-Instruct WebDemo部署.md @@ -1,171 +1,171 @@ -# Qwen2-7B-Instruct WebDemo部署 - -## 环境准备 - -在 Autodl 平台中租赁一个 RTX 3090/24G 显存的显卡机器。如下图所示,镜像选择 PyTorch-->2.1.0-->3.10(ubuntu20.04)-->12.1(11.3 版本以上的都可以)。 - -![03-0](images/03-0.png) - -![03-1](images/03-1.png) - -接下来,我们打开刚刚租用服务器的 JupyterLab,如下图所示,然后打开其中的终端,开始环境配置、模型下载和运行演示。 - -![03-2](images/03-2.png) - -![03-3](images/03-3.png) - -pip 换源加速下载并安装依赖包 - -``` -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install modelscope==1.9.5 -pip install "transformers>=4.37.0" -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.24.1 -pip install transformers_stream_generator==0.0.4 -``` - -> 考虑到部分同学配置环境可能会遇到一些问题,我们在AutoDL平台准备了Qwen2的环境镜像,该镜像适用于该仓库除Qwen-GPTQ和vllm外的所有部署环境。点击下方链接并直接创建Autodl示例即可。 -> ***https://www.codewithgpu.com/i/datawhalechina/self-llm/Qwen2*** - -部署好后的终端如下 - -![03-4](images/03-4.png) - -![03-5](images/03-5.png) - -## 模型下载 - -使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。 - -![03-6](images/03-6.png) - -![03-7](images/03-7.png) - -![03-8](images/03-8.png) - -download.py代码如下 - -``` -import torch -from modelscope import snapshot_download, AutoModel, AutoTokenizer -from modelscope import GenerationConfig -model_dir = snapshot_download('qwen/Qwen2-7B-Chat', cache_dir='/root/autodl-tmp', revision='master') -``` - -保存好后在终端运行 python /root/autodl-tmp/download.py 执行下载,下载模型需要一些时间。 - -``` -python /root/autodl-tmp/download.py -``` - -![03-9](images/03-9.png) - -![03-10](images/03-10.png) - -## 代码准备 - -在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 - -![03-11](images/03-11.png) - -![03-12](images/03-12.png) - -chatBot.py代码如下 - -``` -# 导入所需的库 -from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig -import torch -import streamlit as st - -# 在侧边栏中创建一个标题和一个链接 -with st.sidebar: - st.markdown("## Qwen2 LLM") - "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" - # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 - max_length = st.slider("max_length", 0, 1024, 512, step=1) - -# 创建一个标题和一个副标题 -st.title("💬 Qwen2 Chatbot") -st.caption("🚀 A streamlit chatbot powered by Self-LLM") - -# 定义模型路径 -mode_name_or_path = '/root/autodl-tmp/qwen/Qwen2-7B-Chat' - -# 定义一个函数,用于获取模型和tokenizer -@st.cache_resource -def get_model(): - # 从预训练的模型中获取tokenizer - tokenizer = AutoTokenizer.from_pretrained(mode_name_or_path, use_fast=False) - # 从预训练的模型中获取模型,并设置模型参数 - model = AutoModelForCausalLM.from_pretrained(mode_name_or_path, torch_dtype=torch.bfloat16, device_map="auto") - - return tokenizer, model - -# 加载Qwen2-7B-Chat的model和tokenizer -tokenizer, model = get_model() - -# 如果session_state中没有"messages",则创建一个包含默认消息的列表 -if "messages" not in st.session_state: - st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] - -# 遍历session_state中的所有消息,并显示在聊天界面上 -for msg in st.session_state.messages: - st.chat_message(msg["role"]).write(msg["content"]) - -# 如果用户在聊天输入框中输入了内容,则执行以下操作 -if prompt := st.chat_input(): - # 将用户的输入添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "user", "content": prompt}) - # 在聊天界面上显示用户的输入 - st.chat_message("user").write(prompt) - - # 构建输入 - input_ids = tokenizer.apply_chat_template(st.session_state.messages,tokenize=False,add_generation_prompt=True) - model_inputs = tokenizer([input_ids], return_tensors="pt").to('cuda') - generated_ids = model.generate(model_inputs.input_ids, max_new_tokens=512) - generated_ids = [ - output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) - ] - response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0] - # 将模型的输出添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "assistant", "content": response}) - # 在聊天界面上显示模型的输出 - st.chat_message("assistant").write(response) - # print(st.session_state) -``` - -## 运行demo - -在终端中运行以下命令,启动streamlit服务 - -``` -streamlit run /root/autodl-tmp/chatBot.py --server.address 127.0.0.1 --server.port 6006 -``` - -点击自定义服务 - -![image-20240607213511771](images/03-13.png) - -点开linux - -![image-20240607213618838](images/03-14.png) - -然后win+R打开powershell - -![image-20240607213655624](images/03-15.png) - -输入ssh与密码,按下回车至这样即可 - -![image-20240607213844040](images/03-16.png) - -在浏览器中打开链接 http://localhost:6006/ ,即可看到聊天界面。运行效果如下:![03-13](images/03-17.png) - +# Qwen2-7B-Instruct WebDemo部署 + +## 环境准备 + +在 Autodl 平台中租赁一个 RTX 3090/24G 显存的显卡机器。如下图所示,镜像选择 PyTorch-->2.1.0-->3.10(ubuntu20.04)-->12.1(11.3 版本以上的都可以)。 + +![03-0](images/03-0.png) + +![03-1](images/03-1.png) + +接下来,我们打开刚刚租用服务器的 JupyterLab,如下图所示,然后打开其中的终端,开始环境配置、模型下载和运行演示。 + +![03-2](images/03-2.png) + +![03-3](images/03-3.png) + +pip 换源加速下载并安装依赖包 + +``` +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install modelscope==1.9.5 +pip install "transformers>=4.37.0" +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.24.1 +pip install transformers_stream_generator==0.0.4 +``` + +> 考虑到部分同学配置环境可能会遇到一些问题,我们在AutoDL平台准备了Qwen2的环境镜像,该镜像适用于该仓库除Qwen-GPTQ和vllm外的所有部署环境。点击下方链接并直接创建Autodl示例即可。 +> ***https://www.codewithgpu.com/i/datawhalechina/self-llm/Qwen2*** + +部署好后的终端如下 + +![03-4](images/03-4.png) + +![03-5](images/03-5.png) + +## 模型下载 + +使用 modelscope 中的snapshot_download函数下载模型,第一个参数为模型名称,参数cache_dir为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。 + +![03-6](images/03-6.png) + +![03-7](images/03-7.png) + +![03-8](images/03-8.png) + +download.py代码如下 + +``` +import torch +from modelscope import snapshot_download, AutoModel, AutoTokenizer +from modelscope import GenerationConfig +model_dir = snapshot_download('qwen/Qwen2-7B-Chat', cache_dir='/root/autodl-tmp', revision='master') +``` + +保存好后在终端运行 python /root/autodl-tmp/download.py 执行下载,下载模型需要一些时间。 + +``` +python /root/autodl-tmp/download.py +``` + +![03-9](images/03-9.png) + +![03-10](images/03-10.png) + +## 代码准备 + +在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 + +![03-11](images/03-11.png) + +![03-12](images/03-12.png) + +chatBot.py代码如下 + +``` +# 导入所需的库 +from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig +import torch +import streamlit as st + +# 在侧边栏中创建一个标题和一个链接 +with st.sidebar: + st.markdown("## Qwen2 LLM") + "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" + # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 + max_length = st.slider("max_length", 0, 1024, 512, step=1) + +# 创建一个标题和一个副标题 +st.title("💬 Qwen2 Chatbot") +st.caption("🚀 A streamlit chatbot powered by Self-LLM") + +# 定义模型路径 +mode_name_or_path = '/root/autodl-tmp/qwen/Qwen2-7B-Chat' + +# 定义一个函数,用于获取模型和tokenizer +@st.cache_resource +def get_model(): + # 从预训练的模型中获取tokenizer + tokenizer = AutoTokenizer.from_pretrained(mode_name_or_path, use_fast=False) + # 从预训练的模型中获取模型,并设置模型参数 + model = AutoModelForCausalLM.from_pretrained(mode_name_or_path, torch_dtype=torch.bfloat16, device_map="auto") + + return tokenizer, model + +# 加载Qwen2-7B-Chat的model和tokenizer +tokenizer, model = get_model() + +# 如果session_state中没有"messages",则创建一个包含默认消息的列表 +if "messages" not in st.session_state: + st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] + +# 遍历session_state中的所有消息,并显示在聊天界面上 +for msg in st.session_state.messages: + st.chat_message(msg["role"]).write(msg["content"]) + +# 如果用户在聊天输入框中输入了内容,则执行以下操作 +if prompt := st.chat_input(): + # 将用户的输入添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "user", "content": prompt}) + # 在聊天界面上显示用户的输入 + st.chat_message("user").write(prompt) + + # 构建输入 + input_ids = tokenizer.apply_chat_template(st.session_state.messages,tokenize=False,add_generation_prompt=True) + model_inputs = tokenizer([input_ids], return_tensors="pt").to('cuda') + generated_ids = model.generate(model_inputs.input_ids, max_new_tokens=512) + generated_ids = [ + output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) + ] + response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0] + # 将模型的输出添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "assistant", "content": response}) + # 在聊天界面上显示模型的输出 + st.chat_message("assistant").write(response) + # print(st.session_state) +``` + +## 运行demo + +在终端中运行以下命令,启动streamlit服务 + +``` +streamlit run /root/autodl-tmp/chatBot.py --server.address 127.0.0.1 --server.port 6006 +``` + +点击自定义服务 + +![image-20240607213511771](images/03-13.png) + +点开linux + +![image-20240607213618838](images/03-14.png) + +然后win+R打开powershell + +![image-20240607213655624](images/03-15.png) + +输入ssh与密码,按下回车至这样即可 + +![image-20240607213844040](images/03-16.png) + +在浏览器中打开链接 http://localhost:6006/ ,即可看到聊天界面。运行效果如下:![03-13](images/03-17.png) + diff --git a/Qwen2/04-Qwen2-7B-Instruct vLLM 部署调用.md b/models/Qwen2/04-Qwen2-7B-Instruct vLLM 部署调用.md similarity index 100% rename from Qwen2/04-Qwen2-7B-Instruct vLLM 部署调用.md rename to models/Qwen2/04-Qwen2-7B-Instruct vLLM 部署调用.md diff --git a/Qwen2/05-Qwen2-7B-Instruct Lora 微调.md b/models/Qwen2/05-Qwen2-7B-Instruct Lora 微调.md similarity index 100% rename from Qwen2/05-Qwen2-7B-Instruct Lora 微调.md rename to models/Qwen2/05-Qwen2-7B-Instruct Lora 微调.md diff --git a/Qwen2/05-Qwen2-7B-Instruct Lora.ipynb b/models/Qwen2/05-Qwen2-7B-Instruct Lora.ipynb similarity index 100% rename from Qwen2/05-Qwen2-7B-Instruct Lora.ipynb rename to models/Qwen2/05-Qwen2-7B-Instruct Lora.ipynb diff --git a/Qwen2/benchmark_throughput.py b/models/Qwen2/benchmark_throughput.py similarity index 100% rename from Qwen2/benchmark_throughput.py rename to models/Qwen2/benchmark_throughput.py diff --git a/Qwen2/images/01-0.png b/models/Qwen2/images/01-0.png similarity index 100% rename from Qwen2/images/01-0.png rename to models/Qwen2/images/01-0.png diff --git a/Qwen2/images/01-1.png b/models/Qwen2/images/01-1.png similarity index 100% rename from Qwen2/images/01-1.png rename to models/Qwen2/images/01-1.png diff --git a/Qwen2/images/01-2.png b/models/Qwen2/images/01-2.png similarity index 100% rename from Qwen2/images/01-2.png rename to models/Qwen2/images/01-2.png diff --git a/Qwen2/images/01-3.png b/models/Qwen2/images/01-3.png similarity index 100% rename from Qwen2/images/01-3.png rename to models/Qwen2/images/01-3.png diff --git a/Qwen2/images/01-4.png b/models/Qwen2/images/01-4.png similarity index 100% rename from Qwen2/images/01-4.png rename to models/Qwen2/images/01-4.png diff --git a/Qwen2/images/01-5.png b/models/Qwen2/images/01-5.png similarity index 100% rename from Qwen2/images/01-5.png rename to models/Qwen2/images/01-5.png diff --git a/Qwen2/images/01-6.png b/models/Qwen2/images/01-6.png similarity index 100% rename from Qwen2/images/01-6.png rename to models/Qwen2/images/01-6.png diff --git a/Qwen2/images/01-7.png b/models/Qwen2/images/01-7.png similarity index 100% rename from Qwen2/images/01-7.png rename to models/Qwen2/images/01-7.png diff --git a/Qwen2/images/02-1.png b/models/Qwen2/images/02-1.png similarity index 100% rename from Qwen2/images/02-1.png rename to models/Qwen2/images/02-1.png diff --git a/Qwen2/images/03-0.png b/models/Qwen2/images/03-0.png similarity index 100% rename from Qwen2/images/03-0.png rename to models/Qwen2/images/03-0.png diff --git a/Qwen2/images/03-1.png b/models/Qwen2/images/03-1.png similarity index 100% rename from Qwen2/images/03-1.png rename to models/Qwen2/images/03-1.png diff --git a/Qwen2/images/03-10.png b/models/Qwen2/images/03-10.png similarity index 100% rename from Qwen2/images/03-10.png rename to models/Qwen2/images/03-10.png diff --git a/Qwen2/images/03-11.png b/models/Qwen2/images/03-11.png similarity index 100% rename from Qwen2/images/03-11.png rename to models/Qwen2/images/03-11.png diff --git a/Qwen2/images/03-12.png b/models/Qwen2/images/03-12.png similarity index 100% rename from Qwen2/images/03-12.png rename to models/Qwen2/images/03-12.png diff --git a/Qwen2/images/03-13.png b/models/Qwen2/images/03-13.png similarity index 100% rename from Qwen2/images/03-13.png rename to models/Qwen2/images/03-13.png diff --git a/Qwen2/images/03-14.png b/models/Qwen2/images/03-14.png similarity index 100% rename from Qwen2/images/03-14.png rename to models/Qwen2/images/03-14.png diff --git a/Qwen2/images/03-15.png b/models/Qwen2/images/03-15.png similarity index 100% rename from Qwen2/images/03-15.png rename to models/Qwen2/images/03-15.png diff --git a/Qwen2/images/03-16.png b/models/Qwen2/images/03-16.png similarity index 100% rename from Qwen2/images/03-16.png rename to models/Qwen2/images/03-16.png diff --git a/Qwen2/images/03-17.png b/models/Qwen2/images/03-17.png similarity index 100% rename from Qwen2/images/03-17.png rename to models/Qwen2/images/03-17.png diff --git a/Qwen2/images/03-2.png b/models/Qwen2/images/03-2.png similarity index 100% rename from Qwen2/images/03-2.png rename to models/Qwen2/images/03-2.png diff --git a/Qwen2/images/03-3.png b/models/Qwen2/images/03-3.png similarity index 100% rename from Qwen2/images/03-3.png rename to models/Qwen2/images/03-3.png diff --git a/Qwen2/images/03-4.png b/models/Qwen2/images/03-4.png similarity index 100% rename from Qwen2/images/03-4.png rename to models/Qwen2/images/03-4.png diff --git a/Qwen2/images/03-5.png b/models/Qwen2/images/03-5.png similarity index 100% rename from Qwen2/images/03-5.png rename to models/Qwen2/images/03-5.png diff --git a/Qwen2/images/03-6.png b/models/Qwen2/images/03-6.png similarity index 100% rename from Qwen2/images/03-6.png rename to models/Qwen2/images/03-6.png diff --git a/Qwen2/images/03-7.png b/models/Qwen2/images/03-7.png similarity index 100% rename from Qwen2/images/03-7.png rename to models/Qwen2/images/03-7.png diff --git a/Qwen2/images/03-8.png b/models/Qwen2/images/03-8.png similarity index 100% rename from Qwen2/images/03-8.png rename to models/Qwen2/images/03-8.png diff --git a/Qwen2/images/03-9.png b/models/Qwen2/images/03-9.png similarity index 100% rename from Qwen2/images/03-9.png rename to models/Qwen2/images/03-9.png diff --git a/Qwen2/images/fig4-1.png b/models/Qwen2/images/fig4-1.png similarity index 100% rename from Qwen2/images/fig4-1.png rename to models/Qwen2/images/fig4-1.png diff --git a/Qwen2/images/fig4-10.png b/models/Qwen2/images/fig4-10.png similarity index 100% rename from Qwen2/images/fig4-10.png rename to models/Qwen2/images/fig4-10.png diff --git a/Qwen2/images/fig4-11.png b/models/Qwen2/images/fig4-11.png similarity index 100% rename from Qwen2/images/fig4-11.png rename to models/Qwen2/images/fig4-11.png diff --git a/Qwen2/images/fig4-12.png b/models/Qwen2/images/fig4-12.png similarity index 100% rename from Qwen2/images/fig4-12.png rename to models/Qwen2/images/fig4-12.png diff --git a/Qwen2/images/fig4-13.png b/models/Qwen2/images/fig4-13.png similarity index 100% rename from Qwen2/images/fig4-13.png rename to models/Qwen2/images/fig4-13.png diff --git a/Qwen2/images/fig4-14.png b/models/Qwen2/images/fig4-14.png similarity index 100% rename from Qwen2/images/fig4-14.png rename to models/Qwen2/images/fig4-14.png diff --git a/Qwen2/images/fig4-2.png b/models/Qwen2/images/fig4-2.png similarity index 100% rename from Qwen2/images/fig4-2.png rename to models/Qwen2/images/fig4-2.png diff --git a/Qwen2/images/fig4-3.png b/models/Qwen2/images/fig4-3.png similarity index 100% rename from Qwen2/images/fig4-3.png rename to models/Qwen2/images/fig4-3.png diff --git a/Qwen2/images/fig4-4.png b/models/Qwen2/images/fig4-4.png similarity index 100% rename from Qwen2/images/fig4-4.png rename to models/Qwen2/images/fig4-4.png diff --git a/Qwen2/images/fig4-5.png b/models/Qwen2/images/fig4-5.png similarity index 100% rename from Qwen2/images/fig4-5.png rename to models/Qwen2/images/fig4-5.png diff --git a/Qwen2/images/fig4-6.png b/models/Qwen2/images/fig4-6.png similarity index 100% rename from Qwen2/images/fig4-6.png rename to models/Qwen2/images/fig4-6.png diff --git a/Qwen2/images/fig4-7.png b/models/Qwen2/images/fig4-7.png similarity index 100% rename from Qwen2/images/fig4-7.png rename to models/Qwen2/images/fig4-7.png diff --git a/Qwen2/images/fig4-8.png b/models/Qwen2/images/fig4-8.png similarity index 100% rename from Qwen2/images/fig4-8.png rename to models/Qwen2/images/fig4-8.png diff --git a/Qwen2/images/fig4-9.png b/models/Qwen2/images/fig4-9.png similarity index 100% rename from Qwen2/images/fig4-9.png rename to models/Qwen2/images/fig4-9.png diff --git a/TransNormerLLM/01-TransNormerLLM-7B FastApi 部署调用.md b/models/TransNormerLLM/01-TransNormerLLM-7B FastApi 部署调用.md similarity index 100% rename from TransNormerLLM/01-TransNormerLLM-7B FastApi 部署调用.md rename to models/TransNormerLLM/01-TransNormerLLM-7B FastApi 部署调用.md diff --git a/TransNormerLLM/02-TransNormerLLM-7B 接入langchain搭建知识库助手.md b/models/TransNormerLLM/02-TransNormerLLM-7B 接入langchain搭建知识库助手.md similarity index 100% rename from TransNormerLLM/02-TransNormerLLM-7B 接入langchain搭建知识库助手.md rename to models/TransNormerLLM/02-TransNormerLLM-7B 接入langchain搭建知识库助手.md diff --git a/TransNormerLLM/03-TransNormerLLM-7B WebDemo.md b/models/TransNormerLLM/03-TransNormerLLM-7B WebDemo.md similarity index 100% rename from TransNormerLLM/03-TransNormerLLM-7B WebDemo.md rename to models/TransNormerLLM/03-TransNormerLLM-7B WebDemo.md diff --git a/TransNormerLLM/04-TransNormerLLM-7B-chat-Lora.ipynb b/models/TransNormerLLM/04-TransNormerLLM-7B-chat-Lora.ipynb similarity index 100% rename from TransNormerLLM/04-TransNormerLLM-7B-chat-Lora.ipynb rename to models/TransNormerLLM/04-TransNormerLLM-7B-chat-Lora.ipynb diff --git a/TransNormerLLM/04-TrasnNormerLLM-7B Lora 微调.md b/models/TransNormerLLM/04-TrasnNormerLLM-7B Lora 微调.md similarity index 100% rename from TransNormerLLM/04-TrasnNormerLLM-7B Lora 微调.md rename to models/TransNormerLLM/04-TrasnNormerLLM-7B Lora 微调.md diff --git a/TransNormerLLM/images/Jupyter-response.png b/models/TransNormerLLM/images/Jupyter-response.png similarity index 100% rename from TransNormerLLM/images/Jupyter-response.png rename to models/TransNormerLLM/images/Jupyter-response.png diff --git a/TransNormerLLM/images/Machine-Config.png b/models/TransNormerLLM/images/Machine-Config.png similarity index 100% rename from TransNormerLLM/images/Machine-Config.png rename to models/TransNormerLLM/images/Machine-Config.png diff --git a/TransNormerLLM/images/TransNormer-structure.png b/models/TransNormerLLM/images/TransNormer-structure.png similarity index 100% rename from TransNormerLLM/images/TransNormer-structure.png rename to models/TransNormerLLM/images/TransNormer-structure.png diff --git a/TransNormerLLM/images/python-terminal.png b/models/TransNormerLLM/images/python-terminal.png similarity index 100% rename from TransNormerLLM/images/python-terminal.png rename to models/TransNormerLLM/images/python-terminal.png diff --git a/TransNormerLLM/images/python-terminal2.png b/models/TransNormerLLM/images/python-terminal2.png similarity index 100% rename from TransNormerLLM/images/python-terminal2.png rename to models/TransNormerLLM/images/python-terminal2.png diff --git a/TransNormerLLM/images/question_to_the_TransNormer.png b/models/TransNormerLLM/images/question_to_the_TransNormer.png similarity index 100% rename from TransNormerLLM/images/question_to_the_TransNormer.png rename to models/TransNormerLLM/images/question_to_the_TransNormer.png diff --git a/TransNormerLLM/images/response.png b/models/TransNormerLLM/images/response.png similarity index 100% rename from TransNormerLLM/images/response.png rename to models/TransNormerLLM/images/response.png diff --git a/TransNormerLLM/images/server-ok.png b/models/TransNormerLLM/images/server-ok.png similarity index 100% rename from TransNormerLLM/images/server-ok.png rename to models/TransNormerLLM/images/server-ok.png diff --git a/TransNormerLLM/images/start-jupyter.png b/models/TransNormerLLM/images/start-jupyter.png similarity index 100% rename from TransNormerLLM/images/start-jupyter.png rename to models/TransNormerLLM/images/start-jupyter.png diff --git a/XVERSE/01-XVERSE-7B-chat Transformers推理.md b/models/XVERSE/01-XVERSE-7B-chat Transformers推理.md similarity index 100% rename from XVERSE/01-XVERSE-7B-chat Transformers推理.md rename to models/XVERSE/01-XVERSE-7B-chat Transformers推理.md diff --git a/XVERSE/02-XVERSE-7B-chat FastAPI部署.md b/models/XVERSE/02-XVERSE-7B-chat FastAPI部署.md similarity index 100% rename from XVERSE/02-XVERSE-7B-chat FastAPI部署.md rename to models/XVERSE/02-XVERSE-7B-chat FastAPI部署.md diff --git a/XVERSE/03-XVERSE-7B-chat langchain 接入.md b/models/XVERSE/03-XVERSE-7B-chat langchain 接入.md similarity index 100% rename from XVERSE/03-XVERSE-7B-chat langchain 接入.md rename to models/XVERSE/03-XVERSE-7B-chat langchain 接入.md diff --git a/XVERSE/04-XVERSE-7B-chat WebDemo 部署.md b/models/XVERSE/04-XVERSE-7B-chat WebDemo 部署.md similarity index 100% rename from XVERSE/04-XVERSE-7B-chat WebDemo 部署.md rename to models/XVERSE/04-XVERSE-7B-chat WebDemo 部署.md diff --git a/XVERSE/05-XVERSE-7B-Chat Lora 微调.ipynb b/models/XVERSE/05-XVERSE-7B-Chat Lora 微调.ipynb similarity index 100% rename from XVERSE/05-XVERSE-7B-Chat Lora 微调.ipynb rename to models/XVERSE/05-XVERSE-7B-Chat Lora 微调.ipynb diff --git a/XVERSE/05-XVERSE-7B-Chat Lora 微调.md b/models/XVERSE/05-XVERSE-7B-Chat Lora 微调.md similarity index 100% rename from XVERSE/05-XVERSE-7B-Chat Lora 微调.md rename to models/XVERSE/05-XVERSE-7B-Chat Lora 微调.md diff --git a/XVERSE/06-XVERSE-MoE-A4.2B.md b/models/XVERSE/06-XVERSE-MoE-A4.2B.md similarity index 100% rename from XVERSE/06-XVERSE-MoE-A4.2B.md rename to models/XVERSE/06-XVERSE-MoE-A4.2B.md diff --git a/XVERSE/code/LLM.py b/models/XVERSE/code/LLM.py similarity index 100% rename from XVERSE/code/LLM.py rename to models/XVERSE/code/LLM.py diff --git a/XVERSE/code/api.py b/models/XVERSE/code/api.py similarity index 100% rename from XVERSE/code/api.py rename to models/XVERSE/code/api.py diff --git a/XVERSE/code/chatBot.py b/models/XVERSE/code/chatBot.py similarity index 100% rename from XVERSE/code/chatBot.py rename to models/XVERSE/code/chatBot.py diff --git a/XVERSE/code/data_format.py b/models/XVERSE/code/data_format.py similarity index 100% rename from XVERSE/code/data_format.py rename to models/XVERSE/code/data_format.py diff --git a/XVERSE/code/model_download.py b/models/XVERSE/code/model_download.py similarity index 100% rename from XVERSE/code/model_download.py rename to models/XVERSE/code/model_download.py diff --git a/XVERSE/code/requirement.txt b/models/XVERSE/code/requirement.txt similarity index 100% rename from XVERSE/code/requirement.txt rename to models/XVERSE/code/requirement.txt diff --git a/XVERSE/code/xverse.py b/models/XVERSE/code/xverse.py similarity index 100% rename from XVERSE/code/xverse.py rename to models/XVERSE/code/xverse.py diff --git a/XVERSE/images/1.png b/models/XVERSE/images/1.png similarity index 100% rename from XVERSE/images/1.png rename to models/XVERSE/images/1.png diff --git a/XVERSE/images/2.png b/models/XVERSE/images/2.png similarity index 100% rename from XVERSE/images/2.png rename to models/XVERSE/images/2.png diff --git a/XVERSE/images/3.png b/models/XVERSE/images/3.png similarity index 100% rename from XVERSE/images/3.png rename to models/XVERSE/images/3.png diff --git a/XVERSE/images/4.png b/models/XVERSE/images/4.png similarity index 100% rename from XVERSE/images/4.png rename to models/XVERSE/images/4.png diff --git a/XVERSE/images/5.png b/models/XVERSE/images/5.png similarity index 100% rename from XVERSE/images/5.png rename to models/XVERSE/images/5.png diff --git a/XVERSE/images/6.png b/models/XVERSE/images/6.png similarity index 100% rename from XVERSE/images/6.png rename to models/XVERSE/images/6.png diff --git a/Yi/01-Yi-6B-Chat FastApi 部署调用.md b/models/Yi/01-Yi-6B-Chat FastApi 部署调用.md similarity index 97% rename from Yi/01-Yi-6B-Chat FastApi 部署调用.md rename to models/Yi/01-Yi-6B-Chat FastApi 部署调用.md index 9059102..1a47a2e 100644 --- a/Yi/01-Yi-6B-Chat FastApi 部署调用.md +++ b/models/Yi/01-Yi-6B-Chat FastApi 部署调用.md @@ -1,162 +1,162 @@ -# Yi-6B-Chat FastApi 部署调用 - -## 环境准备 - -在 Autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8(11.3 版本以上的都可以)。 -接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 - -![开启机器配置选择](images/4.png) - -pip 换源加速下载并安装依赖包 - -```shell -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install fastapi==0.104.1 -pip install uvicorn==0.24.0.post1 -pip install requests==2.25.1 -pip install modelscope==1.9.5 -pip install transformers==4.35.2 -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.24.1 -pip install transformers_stream_generator==0.0.4 -``` - -## 模型下载 - -使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建 model_download.py 文件并在其中输入以下内容,粘贴代码后请及时保存文件,如下图所示。并运行 `python /root/autodl-tmp/model_download.py` 执行下载,模型大小为 12GB,下载模型大概需要 8~15 分钟。 - -```python -import torch -from modelscope import snapshot_download, AutoModel, AutoTokenizer -import os -model_dir = snapshot_download('01ai/Yi-6B-Chat', cache_dir='/root/autodl-tmp', revision='master') -``` - -## 代码准备 - -在 /root/autodl-tmp 路径下新建 api.py 文件并在其中输入以下内容,粘贴代码后请及时保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出 issue。 - -```python -from fastapi import FastAPI, Request -from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig -import uvicorn -import json -import datetime -import torch - -# 设置设备参数 -DEVICE = "cuda" # 使用CUDA -DEVICE_ID = "0" # CUDA设备ID,如果未设置则为空 -CUDA_DEVICE = f"{DEVICE}:{DEVICE_ID}" if DEVICE_ID else DEVICE # 组合CUDA设备信息 - -# 清理GPU内存函数 -def torch_gc(): - if torch.cuda.is_available(): # 检查是否可用CUDA - with torch.cuda.device(CUDA_DEVICE): # 指定CUDA设备 - torch.cuda.empty_cache() # 清空CUDA缓存 - torch.cuda.ipc_collect() # 收集CUDA内存碎片 - -# 创建FastAPI应用 -app = FastAPI() - -# 处理POST请求的端点 -@app.post("/") -async def create_item(request: Request): - global model, tokenizer # 声明全局变量以便在函数内部使用模型和分词器 - json_post_raw = await request.json() # 获取POST请求的JSON数据 - json_post = json.dumps(json_post_raw) # 将JSON数据转换为字符串 - json_post_list = json.loads(json_post) # 将字符串转换为Python对象 - prompt = json_post_list.get('prompt') # 获取请求中的提示 - - messages = [ - {"role": "user", "content": prompt} - ] - - # 调用模型进行对话生成 - input_ids = tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') - output_ids = model.generate(input_ids.to('cuda')) - response = tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) - now = datetime.datetime.now() # 获取当前时间 - time = now.strftime("%Y-%m-%d %H:%M:%S") # 格式化时间为字符串 - # 构建响应JSON - answer = { - "response": response, - "status": 200, - "time": time - } - # 构建日志信息 - log = "[" + time + "] " + '", prompt:"' + prompt + '", response:"' + repr(response) + '"' - print(log) # 打印日志 - torch_gc() # 执行GPU内存清理 - return answer # 返回响应 - -# 主函数入口 -if __name__ == '__main__': - # 加载预训练的分词器和模型 - model_name_or_path = 'root/autodl-tmp/01ai/Yi-6B-Chat' - tokenizer = AutoTokenizer.from_pretrained(model_name_or_path, trust_remote_code=True, use_fast=False) - model = AutoModelForCausalLM.from_pretrained(model_name_or_path, device_map="auto", torch_dtype=torch.bfloat16, trust_remote_code=True).eval() - model.generation_config = GenerationConfig.from_pretrained(model_name_or_path, trust_remote_code=True) # 可指定 - model.eval() # 设置模型为评估模式 - # 启动FastAPI应用 - # 用6006端口可以将autodl的端口映射到本地,从而在本地使用api - uvicorn.run(app, host='0.0.0.0', port=6006, workers=1) # 在指定端口和主机上启动应用 -``` - -## Api 部署 - -在终端输入以下命令启动api服务: - -```shell -cd /root/autodl-tmp -python api.py -``` - -加载完毕后出现如下信息说明成功。 - -![Alt text](images/5.png) - - -默认部署在 6006 端口,通过 POST 方法进行调用,可以使用 curl 调用,如下所示: - -```shell -curl -X POST "http://127.0.0.1:6006" \ - -H 'Content-Type: application/json' \ - -d '{"prompt": "你好", "history": []}' -``` - -也可以使用 python 中的 requests 库进行调用,如下所示: - -```python -import requests -import json - -def get_completion(prompt): - headers = {'Content-Type': 'application/json'} - data = {"prompt": prompt} - response = requests.post(url='http://127.0.0.1:6006', headers=headers, data=json.dumps(data)) - return response.json()['response'] - -if __name__ == '__main__': - print(get_completion('你好')) -``` - -得到的返回值如下所示: - -```json -{ - "response":"你好!有什么可以帮助你的吗?", - "history":[["你好","你好!有什么可以帮助你的吗?"]], - "status":200, - "time":"2023-12-15 20:08:40" -} -``` - -![Alt text](images/6.png) +# Yi-6B-Chat FastApi 部署调用 + +## 环境准备 + +在 Autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8(11.3 版本以上的都可以)。 +接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 + +![开启机器配置选择](images/4.png) + +pip 换源加速下载并安装依赖包 + +```shell +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install fastapi==0.104.1 +pip install uvicorn==0.24.0.post1 +pip install requests==2.25.1 +pip install modelscope==1.9.5 +pip install transformers==4.35.2 +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.24.1 +pip install transformers_stream_generator==0.0.4 +``` + +## 模型下载 + +使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建 model_download.py 文件并在其中输入以下内容,粘贴代码后请及时保存文件,如下图所示。并运行 `python /root/autodl-tmp/model_download.py` 执行下载,模型大小为 12GB,下载模型大概需要 8~15 分钟。 + +```python +import torch +from modelscope import snapshot_download, AutoModel, AutoTokenizer +import os +model_dir = snapshot_download('01ai/Yi-6B-Chat', cache_dir='/root/autodl-tmp', revision='master') +``` + +## 代码准备 + +在 /root/autodl-tmp 路径下新建 api.py 文件并在其中输入以下内容,粘贴代码后请及时保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出 issue。 + +```python +from fastapi import FastAPI, Request +from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig +import uvicorn +import json +import datetime +import torch + +# 设置设备参数 +DEVICE = "cuda" # 使用CUDA +DEVICE_ID = "0" # CUDA设备ID,如果未设置则为空 +CUDA_DEVICE = f"{DEVICE}:{DEVICE_ID}" if DEVICE_ID else DEVICE # 组合CUDA设备信息 + +# 清理GPU内存函数 +def torch_gc(): + if torch.cuda.is_available(): # 检查是否可用CUDA + with torch.cuda.device(CUDA_DEVICE): # 指定CUDA设备 + torch.cuda.empty_cache() # 清空CUDA缓存 + torch.cuda.ipc_collect() # 收集CUDA内存碎片 + +# 创建FastAPI应用 +app = FastAPI() + +# 处理POST请求的端点 +@app.post("/") +async def create_item(request: Request): + global model, tokenizer # 声明全局变量以便在函数内部使用模型和分词器 + json_post_raw = await request.json() # 获取POST请求的JSON数据 + json_post = json.dumps(json_post_raw) # 将JSON数据转换为字符串 + json_post_list = json.loads(json_post) # 将字符串转换为Python对象 + prompt = json_post_list.get('prompt') # 获取请求中的提示 + + messages = [ + {"role": "user", "content": prompt} + ] + + # 调用模型进行对话生成 + input_ids = tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') + output_ids = model.generate(input_ids.to('cuda')) + response = tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) + now = datetime.datetime.now() # 获取当前时间 + time = now.strftime("%Y-%m-%d %H:%M:%S") # 格式化时间为字符串 + # 构建响应JSON + answer = { + "response": response, + "status": 200, + "time": time + } + # 构建日志信息 + log = "[" + time + "] " + '", prompt:"' + prompt + '", response:"' + repr(response) + '"' + print(log) # 打印日志 + torch_gc() # 执行GPU内存清理 + return answer # 返回响应 + +# 主函数入口 +if __name__ == '__main__': + # 加载预训练的分词器和模型 + model_name_or_path = 'root/autodl-tmp/01ai/Yi-6B-Chat' + tokenizer = AutoTokenizer.from_pretrained(model_name_or_path, trust_remote_code=True, use_fast=False) + model = AutoModelForCausalLM.from_pretrained(model_name_or_path, device_map="auto", torch_dtype=torch.bfloat16, trust_remote_code=True).eval() + model.generation_config = GenerationConfig.from_pretrained(model_name_or_path, trust_remote_code=True) # 可指定 + model.eval() # 设置模型为评估模式 + # 启动FastAPI应用 + # 用6006端口可以将autodl的端口映射到本地,从而在本地使用api + uvicorn.run(app, host='0.0.0.0', port=6006, workers=1) # 在指定端口和主机上启动应用 +``` + +## Api 部署 + +在终端输入以下命令启动api服务: + +```shell +cd /root/autodl-tmp +python api.py +``` + +加载完毕后出现如下信息说明成功。 + +![Alt text](images/5.png) + + +默认部署在 6006 端口,通过 POST 方法进行调用,可以使用 curl 调用,如下所示: + +```shell +curl -X POST "http://127.0.0.1:6006" \ + -H 'Content-Type: application/json' \ + -d '{"prompt": "你好", "history": []}' +``` + +也可以使用 python 中的 requests 库进行调用,如下所示: + +```python +import requests +import json + +def get_completion(prompt): + headers = {'Content-Type': 'application/json'} + data = {"prompt": prompt} + response = requests.post(url='http://127.0.0.1:6006', headers=headers, data=json.dumps(data)) + return response.json()['response'] + +if __name__ == '__main__': + print(get_completion('你好')) +``` + +得到的返回值如下所示: + +```json +{ + "response":"你好!有什么可以帮助你的吗?", + "history":[["你好","你好!有什么可以帮助你的吗?"]], + "status":200, + "time":"2023-12-15 20:08:40" +} +``` + +![Alt text](images/6.png) diff --git a/Yi/02-Yi-6B-Chat 接入langchain搭建知识库助手.md b/models/Yi/02-Yi-6B-Chat 接入langchain搭建知识库助手.md similarity index 97% rename from Yi/02-Yi-6B-Chat 接入langchain搭建知识库助手.md rename to models/Yi/02-Yi-6B-Chat 接入langchain搭建知识库助手.md index d0bf64a..6935e73 100644 --- a/Yi/02-Yi-6B-Chat 接入langchain搭建知识库助手.md +++ b/models/Yi/02-Yi-6B-Chat 接入langchain搭建知识库助手.md @@ -1,378 +1,378 @@ -# Yi-6B-Chat 接入 LangChain 搭建知识库助手 - -## 环境准备 - -在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 - -![机器配置选择](images/4.png) -接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行 demo。 - -pip 换源加速下载并安装依赖包 - -```shell -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install modelscope==1.9.5 -pip install "transformers>=4.32.0" accelerate tiktoken einops scipy transformers_stream_generator==0.0.4 peft deepspeed -pip install -U huggingface_hub -``` - -## 模型下载 - -在已完成 Yi-6B-chat 部署的基础上,我们还需要还需要安装以下依赖包。 -请在终端复制粘贴以下命令,并回车运行: - -```shell -pip install langchain==0.0.292 -pip install gradio==4.4.0 -pip install chromadb==0.4.15 -pip install sentence-transformers==2.2.2 -pip install unstructured==0.10.30 -pip install markdown==3.3.7 -``` - -同时,我们还需要使用到开源词向量模型 [Sentence Transformer](https://huggingface.co/sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2) 。 - -这里使用 huggingface 镜像下载到本地 /root/autodl-tmp/embedding_model,你也可以选择其它的方式下载。 - -在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后请及时保存文件,如下图所示。并运行 `python /root/autodl-tmp/download.py` 执行下载。 - -```python -import os -# 设置环境变量 -os.environ['HF_ENDPOINT'] = 'https://hf-mirror.com' -# 下载模型 -os.system('huggingface-cli download --resume-download sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2 --local-dir /root/autodl-tmp/embedding_model') -``` - -使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建 model_download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 `python /root/autodl-tmp/model_download.py` 执行下载,模型大小为 11 GB,下载模型大概需要 8~15 分钟。 - -```python - -import torch -from modelscope import snapshot_download, AutoModel, AutoTokenizer -import os -model_dir = snapshot_download('01ai/Yi-6B-Chat', cache_dir='/root/autodl-tmp', revision='master') -``` - -## 知识库建设 - -我们选用以下开源仓库作为知识库来源: - -- [sweettalk-django4.2](https://github.com/Joe-2002/sweettalk-django4.2) - -首先我们需要将上述远程开源仓库 Clone 到本地,可以使用以下命令: - -```shell -# 进入到数据库盘 -cd /root/autodl-tmp -# 打开学术资源加速 -source /etc/network_turbo -# clone 开源仓库 -git clone https://github.com/Joe-2002/sweettalk-django4.2.git -# 关闭学术资源加速 -unset http_proxy && unset https_proxy -``` - -接着,为语料处理方便,我们将选用上述仓库中所有的 markdown、txt 文件作为示例语料库。注意,也可以选用其中的代码文件加入到知识库中,但需要针对代码文件格式进行额外处理。 - -我们首先将上述仓库中所有满足条件的文件路径找出来,我们定义一个函数,该函数将递归指定文件夹路径,返回其中所有满足条件(即后缀名为 .md 或者 .txt 的文件)的文件路径: - -```python -import os -def get_files(dir_path): - # args:dir_path,目标文件夹路径 - file_list = [] - for filepath, dirnames, filenames in os.walk(dir_path): - # os.walk 函数将递归遍历指定文件夹 - for filename in filenames: - # 通过后缀名判断文件类型是否满足要求 - if filename.endswith(".md"): - # 如果满足要求,将其绝对路径加入到结果列表 - file_list.append(os.path.join(filepath, filename)) - elif filename.endswith(".txt"): - file_list.append(os.path.join(filepath, filename)) - return file_list -``` - -得到所有目标文件路径之后,我们可以使用 LangChain 提供的 FileLoader 对象来加载目标文件,得到由目标文件解析出的纯文本内容。由于不同类型的文件需要对应不同的 FileLoader,我们判断目标文件类型,并针对性调用对应类型的 FileLoader,同时,调用 FileLoader 对象的 load 方法来得到加载之后的纯文本对象: - -```python -from tqdm import tqdm -from langchain.document_loaders import UnstructuredFileLoader -from langchain.document_loaders import UnstructuredMarkdownLoader - -def get_text(dir_path): - # args:dir_path,目标文件夹路径 - # 首先调用上文定义的函数得到目标文件路径列表 - file_lst = get_files(dir_path) - # docs 存放加载之后的纯文本对象 - docs = [] - # 遍历所有目标文件 - for one_file in tqdm(file_lst): - file_type = one_file.split('.')[-1] - if file_type == 'md': - loader = UnstructuredMarkdownLoader(one_file) - elif file_type == 'txt': - loader = UnstructuredFileLoader(one_file) - else: - # 如果是不符合条件的文件,直接跳过 - continue - docs.extend(loader.load()) - return docs -``` - -使用上文函数,我们得到的 docs 为一个纯文本对象对应的列表。 - -```python -docs = get_text('/root/autodl-tmp/sweettalk-django4.2') -``` - -得到该列表之后,我们就可以将它引入到 LangChain 框架中构建向量数据库。由纯文本对象构建向量数据库,我们需要先对文本进行分块,接着对文本块进行向量化。 - -LangChain 提供了多种文本分块工具,此处我们使用字符串递归分割器,并选择分块大小为 500,块重叠长度为 150: - -```python -from langchain.text_splitter import RecursiveCharacterTextSplitter - -text_splitter = RecursiveCharacterTextSplitter( - chunk_size=500, chunk_overlap=150) -split_docs = text_splitter.split_documents(docs) -``` - -接着我们选用开源词向量模型 [Sentence Transformer](https://huggingface.co/sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2) 来进行文本向量化。 - -LangChain 提供了直接引入 HuggingFace 开源社区中的模型进行向量化的接口: - -```python -from langchain.embeddings.huggingface import HuggingFaceEmbeddings - -embeddings = HuggingFaceEmbeddings(model_name="/root/autodl-tmp/embedding_model") -``` - -同时,我们选择 Chroma 作为向量数据库,基于上文分块后的文档以及加载的开源向量化模型,将语料加载到指定路径下的向量数据库: - -```python -from langchain.vectorstores import Chroma - -# 定义持久化路径 -persist_directory = 'data_base/vector_db/chroma' -# 加载数据库 -vectordb = Chroma.from_documents( - documents=split_docs, - embedding=embeddings, - persist_directory=persist_directory # 允许我们将persist_directory目录保存到磁盘上 -) -# 将加载的向量数据库持久化到磁盘上 -vectordb.persist() -``` - -将上述代码整合在一起为知识库搭建的脚本: - -```python -# 首先导入所需第三方库 -from langchain.document_loaders import UnstructuredFileLoader -from langchain.document_loaders import UnstructuredMarkdownLoader -from langchain.text_splitter import RecursiveCharacterTextSplitter -from langchain.vectorstores import Chroma -from langchain.embeddings.huggingface import HuggingFaceEmbeddings -from tqdm import tqdm -import os - -# 获取文件路径函数 -def get_files(dir_path): - # args:dir_path,目标文件夹路径 - file_list = [] - for filepath, dirnames, filenames in os.walk(dir_path): - # os.walk 函数将递归遍历指定文件夹 - for filename in filenames: - # 通过后缀名判断文件类型是否满足要求 - if filename.endswith(".md"): - # 如果满足要求,将其绝对路径加入到结果列表 - file_list.append(os.path.join(filepath, filename)) - elif filename.endswith(".txt"): - file_list.append(os.path.join(filepath, filename)) - return file_list - -# 加载文件函数 -def get_text(dir_path): - # args:dir_path,目标文件夹路径 - # 首先调用上文定义的函数得到目标文件路径列表 - file_lst = get_files(dir_path) - # docs 存放加载之后的纯文本对象 - docs = [] - # 遍历所有目标文件 - for one_file in tqdm(file_lst): - file_type = one_file.split('.')[-1] - if file_type == 'md': - loader = UnstructuredMarkdownLoader(one_file) - elif file_type == 'txt': - loader = UnstructuredFileLoader(one_file) - else: - # 如果是不符合条件的文件,直接跳过 - continue - docs.extend(loader.load()) - return docs - -# 目标文件夹 -tar_dir = [ - "/root/autodl-tmp/sweettalk-django4.2", -] - -# 加载目标文件 -docs = [] -for dir_path in tar_dir: - docs.extend(get_text(dir_path)) - -# 对文本进行分块 -text_splitter = RecursiveCharacterTextSplitter( - chunk_size=500, chunk_overlap=150) -split_docs = text_splitter.split_documents(docs) - -# 加载开源词向量模型 -embeddings = HuggingFaceEmbeddings(model_name="/root/autodl-tmp/embedding_model") - -# 构建向量数据库 -# 定义持久化路径 -persist_directory = 'data_base/vector_db/chroma' -# 加载数据库 -vectordb = Chroma.from_documents( - documents=split_docs, - embedding=embeddings, - persist_directory=persist_directory # 允许我们将persist_directory目录保存到磁盘上 -) -# 将加载的向量数据库持久化到磁盘上 -vectordb.persist() -``` - -运行上述脚本,即可在本地构建已持久化的向量数据库,后续直接导入该数据库即可,无需重复构建。 - -## Yi 接入LangChain - -为便捷构建 LLM 应用,我们需要基于本地部署的 YiLM,自定义一个 LLM 类,将 Yi 接入到 LangChain 框架中。完成自定义 LLM 类之后,可以以完全一致的方式调用 LangChain 的接口,而无需考虑底层模型调用的不一致。 - -基于本地部署的 Yi 自定义 LLM 类并不复杂,我们只需从 LangChain.llms.base.LLM 类继承一个子类,并重写构造函数与 _call 函数即可: - -```python -from langchain.llms.base import LLM -from typing import Any, List, Optional -from langchain.callbacks.manager import CallbackManagerForLLMRun -from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig, LlamaTokenizerFast -import torch - -class Yi_LLM(LLM): - # 基于本地 Yi 自定义 LLM 类 - tokenizer: AutoTokenizer = None - model: AutoModelForCausalLM = None - - def __init__(self, mode_name_or_path :str): - - super().__init__() - print("正在从本地加载模型...") - self.tokenizer = AutoTokenizer.from_pretrained(mode_name_or_path, trust_remote_code=True, use_fast=False) - self.model = AutoModelForCausalLM.from_pretrained(mode_name_or_path, trust_remote_code=True,torch_dtype=torch.bfloat16,device_map="auto") - self.model.generation_config = GenerationConfig.from_pretrained(mode_name_or_path) - self.model.generation_config.pad_token_id = self.model.generation_config.eos_token_id - self.model = self.model.eval() - print("完成本地模型的加载") - - def _call(self, prompt : str, stop: Optional[List[str]] = None, - run_manager: Optional[CallbackManagerForLLMRun] = None, - **kwargs: Any): - - messages = [ - {"role": "user", "content": prompt } - ] - input_ids = self.tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') - - output_ids = self.model.generate(input_ids.to('cuda')) - response = self.tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) - return response - @property - def _llm_type(self) -> str: - return "Yi_LLM" -``` - -在上述类定义中,我们分别重写了构造函数和 _call 函数:对于构造函数,我们在对象实例化的一开始加载本地部署的 Yi 模型,从而避免每一次调用都需要重新加载模型带来的时间过长;_call 函数是 LLM 类的核心函数,LangChain 会调用该函数来调用 LLM,在该函数中,我们调用已实例化模型的 generate 方法,从而实现对模型的调用并返回调用结果。 - -在整体项目中,我们将上述代码封装为 LLM.py,后续将直接从该文件中引入自定义的 LLM 类。 - -## 构建检索问答链 - -LangChain 通过提供检索问答链对象来实现对于 RAG 全流程的封装。即我们可以调用一个 LangChain 提供的 RetrievalQA 对象,通过初始化时填入已构建的数据库和自定义 LLM 作为参数,来简便地完成检索增强问答的全流程,LangChain 会自动完成基于用户提问进行检索、获取相关文档、拼接为合适的 Prompt 并交给 LLM 问答的全部流程。 - -首先我们需要将上文构建的向量数据库导入进来,我们可以直接通过 Chroma 以及上文定义的词向量模型来加载已构建的数据库: - -```python -from langchain.vectorstores import Chroma -from langchain.embeddings.huggingface import HuggingFaceEmbeddings -import os - -# 定义 Embeddings -embeddings = HuggingFaceEmbeddings(model_name="/root/autodl-tmp/embedding_model") - -# 向量数据库持久化路径 -persist_directory = 'data_base/vector_db/chroma' - -# 加载数据库 -vectordb = Chroma( - persist_directory=persist_directory, - embedding_function=embeddings -) -``` - -上述代码得到的 vectordb 对象即为我们已构建的向量数据库对象,该对象可以针对用户的 query 进行语义向量检索,得到与用户提问相关的知识片段。 - -接着,我们实例化一个基于 Yi 自定义的 LLM 对象: - -```python -from LLM import Yi_LLM -llm = Yi_LLM(mode_name_or_path = "/root/autodl-tmp/01ai/Yi-6B-Chat") -llm("你是谁") -``` - -![模型返回回答效果](images/question_to_the_Yi.png) -构建检索问答链,还需要构建一个 Prompt Template,该 Template 其实基于一个带变量的字符串,在检索之后,LangChain 会将检索到的相关文档片段填入到 Template 的变量中,从而实现带知识的 Prompt 构建。我们可以基于 LangChain 的 Template 基类来实例化这样一个 Template 对象: - -```python -from langchain.prompts import PromptTemplate - -# 我们所构造的 Prompt 模板 -template = """使用以下上下文来回答最后的问题。如果你不知道答案,就说你不知道,不要试图编造答案。尽量使答案简明扼要。总是在回答的最后说“谢谢你的提问!”。 -{context} -问题: {question} -有用的回答:""" - -# 调用 LangChain 的方法来实例化一个 Template 对象,该对象包含了 context 和 question 两个变量,在实际调用时,这两个变量会被检索到的文档片段和用户提问填充 -QA_CHAIN_PROMPT = PromptTemplate(input_variables=["context","question"],template=template) -``` - -最后,可以调用 LangChain 提供的检索问答链构造函数,基于我们的自定义 LLM、Prompt Template 和向量知识库来构建一个基于 Yi 的检索问答链: - -```python -from langchain.chains import RetrievalQA - -qa_chain = RetrievalQA.from_chain_type(llm,retriever=vectordb.as_retriever(),return_source_documents=True,chain_type_kwargs={"prompt":QA_CHAIN_PROMPT}) -``` - -得到的 qa_chain 对象即可以实现我们的核心功能,即基于 Yi 模型的专业知识库助手。我们可以对比该检索问答链和纯 LLM 的问答效果: - -```python -question = "sweettalk_django项目是什么" -result = qa_chain({"query": question}) -print("检索问答链回答 question 的结果:") -print(result["result"]) - -print("-------------------") -# 仅 LLM 回答效果 -result_2 = llm(question) -print("大模型回答 question 的结果:") -print(result_2) -``` - +# Yi-6B-Chat 接入 LangChain 搭建知识库助手 + +## 环境准备 + +在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 + +![机器配置选择](images/4.png) +接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行 demo。 + +pip 换源加速下载并安装依赖包 + +```shell +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install modelscope==1.9.5 +pip install "transformers>=4.32.0" accelerate tiktoken einops scipy transformers_stream_generator==0.0.4 peft deepspeed +pip install -U huggingface_hub +``` + +## 模型下载 + +在已完成 Yi-6B-chat 部署的基础上,我们还需要还需要安装以下依赖包。 +请在终端复制粘贴以下命令,并回车运行: + +```shell +pip install langchain==0.0.292 +pip install gradio==4.4.0 +pip install chromadb==0.4.15 +pip install sentence-transformers==2.2.2 +pip install unstructured==0.10.30 +pip install markdown==3.3.7 +``` + +同时,我们还需要使用到开源词向量模型 [Sentence Transformer](https://huggingface.co/sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2) 。 + +这里使用 huggingface 镜像下载到本地 /root/autodl-tmp/embedding_model,你也可以选择其它的方式下载。 + +在 /root/autodl-tmp 路径下新建 download.py 文件并在其中输入以下内容,粘贴代码后请及时保存文件,如下图所示。并运行 `python /root/autodl-tmp/download.py` 执行下载。 + +```python +import os +# 设置环境变量 +os.environ['HF_ENDPOINT'] = 'https://hf-mirror.com' +# 下载模型 +os.system('huggingface-cli download --resume-download sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2 --local-dir /root/autodl-tmp/embedding_model') +``` + +使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建 model_download.py 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 `python /root/autodl-tmp/model_download.py` 执行下载,模型大小为 11 GB,下载模型大概需要 8~15 分钟。 + +```python + +import torch +from modelscope import snapshot_download, AutoModel, AutoTokenizer +import os +model_dir = snapshot_download('01ai/Yi-6B-Chat', cache_dir='/root/autodl-tmp', revision='master') +``` + +## 知识库建设 + +我们选用以下开源仓库作为知识库来源: + +- [sweettalk-django4.2](https://github.com/Joe-2002/sweettalk-django4.2) + +首先我们需要将上述远程开源仓库 Clone 到本地,可以使用以下命令: + +```shell +# 进入到数据库盘 +cd /root/autodl-tmp +# 打开学术资源加速 +source /etc/network_turbo +# clone 开源仓库 +git clone https://github.com/Joe-2002/sweettalk-django4.2.git +# 关闭学术资源加速 +unset http_proxy && unset https_proxy +``` + +接着,为语料处理方便,我们将选用上述仓库中所有的 markdown、txt 文件作为示例语料库。注意,也可以选用其中的代码文件加入到知识库中,但需要针对代码文件格式进行额外处理。 + +我们首先将上述仓库中所有满足条件的文件路径找出来,我们定义一个函数,该函数将递归指定文件夹路径,返回其中所有满足条件(即后缀名为 .md 或者 .txt 的文件)的文件路径: + +```python +import os +def get_files(dir_path): + # args:dir_path,目标文件夹路径 + file_list = [] + for filepath, dirnames, filenames in os.walk(dir_path): + # os.walk 函数将递归遍历指定文件夹 + for filename in filenames: + # 通过后缀名判断文件类型是否满足要求 + if filename.endswith(".md"): + # 如果满足要求,将其绝对路径加入到结果列表 + file_list.append(os.path.join(filepath, filename)) + elif filename.endswith(".txt"): + file_list.append(os.path.join(filepath, filename)) + return file_list +``` + +得到所有目标文件路径之后,我们可以使用 LangChain 提供的 FileLoader 对象来加载目标文件,得到由目标文件解析出的纯文本内容。由于不同类型的文件需要对应不同的 FileLoader,我们判断目标文件类型,并针对性调用对应类型的 FileLoader,同时,调用 FileLoader 对象的 load 方法来得到加载之后的纯文本对象: + +```python +from tqdm import tqdm +from langchain.document_loaders import UnstructuredFileLoader +from langchain.document_loaders import UnstructuredMarkdownLoader + +def get_text(dir_path): + # args:dir_path,目标文件夹路径 + # 首先调用上文定义的函数得到目标文件路径列表 + file_lst = get_files(dir_path) + # docs 存放加载之后的纯文本对象 + docs = [] + # 遍历所有目标文件 + for one_file in tqdm(file_lst): + file_type = one_file.split('.')[-1] + if file_type == 'md': + loader = UnstructuredMarkdownLoader(one_file) + elif file_type == 'txt': + loader = UnstructuredFileLoader(one_file) + else: + # 如果是不符合条件的文件,直接跳过 + continue + docs.extend(loader.load()) + return docs +``` + +使用上文函数,我们得到的 docs 为一个纯文本对象对应的列表。 + +```python +docs = get_text('/root/autodl-tmp/sweettalk-django4.2') +``` + +得到该列表之后,我们就可以将它引入到 LangChain 框架中构建向量数据库。由纯文本对象构建向量数据库,我们需要先对文本进行分块,接着对文本块进行向量化。 + +LangChain 提供了多种文本分块工具,此处我们使用字符串递归分割器,并选择分块大小为 500,块重叠长度为 150: + +```python +from langchain.text_splitter import RecursiveCharacterTextSplitter + +text_splitter = RecursiveCharacterTextSplitter( + chunk_size=500, chunk_overlap=150) +split_docs = text_splitter.split_documents(docs) +``` + +接着我们选用开源词向量模型 [Sentence Transformer](https://huggingface.co/sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2) 来进行文本向量化。 + +LangChain 提供了直接引入 HuggingFace 开源社区中的模型进行向量化的接口: + +```python +from langchain.embeddings.huggingface import HuggingFaceEmbeddings + +embeddings = HuggingFaceEmbeddings(model_name="/root/autodl-tmp/embedding_model") +``` + +同时,我们选择 Chroma 作为向量数据库,基于上文分块后的文档以及加载的开源向量化模型,将语料加载到指定路径下的向量数据库: + +```python +from langchain.vectorstores import Chroma + +# 定义持久化路径 +persist_directory = 'data_base/vector_db/chroma' +# 加载数据库 +vectordb = Chroma.from_documents( + documents=split_docs, + embedding=embeddings, + persist_directory=persist_directory # 允许我们将persist_directory目录保存到磁盘上 +) +# 将加载的向量数据库持久化到磁盘上 +vectordb.persist() +``` + +将上述代码整合在一起为知识库搭建的脚本: + +```python +# 首先导入所需第三方库 +from langchain.document_loaders import UnstructuredFileLoader +from langchain.document_loaders import UnstructuredMarkdownLoader +from langchain.text_splitter import RecursiveCharacterTextSplitter +from langchain.vectorstores import Chroma +from langchain.embeddings.huggingface import HuggingFaceEmbeddings +from tqdm import tqdm +import os + +# 获取文件路径函数 +def get_files(dir_path): + # args:dir_path,目标文件夹路径 + file_list = [] + for filepath, dirnames, filenames in os.walk(dir_path): + # os.walk 函数将递归遍历指定文件夹 + for filename in filenames: + # 通过后缀名判断文件类型是否满足要求 + if filename.endswith(".md"): + # 如果满足要求,将其绝对路径加入到结果列表 + file_list.append(os.path.join(filepath, filename)) + elif filename.endswith(".txt"): + file_list.append(os.path.join(filepath, filename)) + return file_list + +# 加载文件函数 +def get_text(dir_path): + # args:dir_path,目标文件夹路径 + # 首先调用上文定义的函数得到目标文件路径列表 + file_lst = get_files(dir_path) + # docs 存放加载之后的纯文本对象 + docs = [] + # 遍历所有目标文件 + for one_file in tqdm(file_lst): + file_type = one_file.split('.')[-1] + if file_type == 'md': + loader = UnstructuredMarkdownLoader(one_file) + elif file_type == 'txt': + loader = UnstructuredFileLoader(one_file) + else: + # 如果是不符合条件的文件,直接跳过 + continue + docs.extend(loader.load()) + return docs + +# 目标文件夹 +tar_dir = [ + "/root/autodl-tmp/sweettalk-django4.2", +] + +# 加载目标文件 +docs = [] +for dir_path in tar_dir: + docs.extend(get_text(dir_path)) + +# 对文本进行分块 +text_splitter = RecursiveCharacterTextSplitter( + chunk_size=500, chunk_overlap=150) +split_docs = text_splitter.split_documents(docs) + +# 加载开源词向量模型 +embeddings = HuggingFaceEmbeddings(model_name="/root/autodl-tmp/embedding_model") + +# 构建向量数据库 +# 定义持久化路径 +persist_directory = 'data_base/vector_db/chroma' +# 加载数据库 +vectordb = Chroma.from_documents( + documents=split_docs, + embedding=embeddings, + persist_directory=persist_directory # 允许我们将persist_directory目录保存到磁盘上 +) +# 将加载的向量数据库持久化到磁盘上 +vectordb.persist() +``` + +运行上述脚本,即可在本地构建已持久化的向量数据库,后续直接导入该数据库即可,无需重复构建。 + +## Yi 接入LangChain + +为便捷构建 LLM 应用,我们需要基于本地部署的 YiLM,自定义一个 LLM 类,将 Yi 接入到 LangChain 框架中。完成自定义 LLM 类之后,可以以完全一致的方式调用 LangChain 的接口,而无需考虑底层模型调用的不一致。 + +基于本地部署的 Yi 自定义 LLM 类并不复杂,我们只需从 LangChain.llms.base.LLM 类继承一个子类,并重写构造函数与 _call 函数即可: + +```python +from langchain.llms.base import LLM +from typing import Any, List, Optional +from langchain.callbacks.manager import CallbackManagerForLLMRun +from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig, LlamaTokenizerFast +import torch + +class Yi_LLM(LLM): + # 基于本地 Yi 自定义 LLM 类 + tokenizer: AutoTokenizer = None + model: AutoModelForCausalLM = None + + def __init__(self, mode_name_or_path :str): + + super().__init__() + print("正在从本地加载模型...") + self.tokenizer = AutoTokenizer.from_pretrained(mode_name_or_path, trust_remote_code=True, use_fast=False) + self.model = AutoModelForCausalLM.from_pretrained(mode_name_or_path, trust_remote_code=True,torch_dtype=torch.bfloat16,device_map="auto") + self.model.generation_config = GenerationConfig.from_pretrained(mode_name_or_path) + self.model.generation_config.pad_token_id = self.model.generation_config.eos_token_id + self.model = self.model.eval() + print("完成本地模型的加载") + + def _call(self, prompt : str, stop: Optional[List[str]] = None, + run_manager: Optional[CallbackManagerForLLMRun] = None, + **kwargs: Any): + + messages = [ + {"role": "user", "content": prompt } + ] + input_ids = self.tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') + + output_ids = self.model.generate(input_ids.to('cuda')) + response = self.tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) + return response + @property + def _llm_type(self) -> str: + return "Yi_LLM" +``` + +在上述类定义中,我们分别重写了构造函数和 _call 函数:对于构造函数,我们在对象实例化的一开始加载本地部署的 Yi 模型,从而避免每一次调用都需要重新加载模型带来的时间过长;_call 函数是 LLM 类的核心函数,LangChain 会调用该函数来调用 LLM,在该函数中,我们调用已实例化模型的 generate 方法,从而实现对模型的调用并返回调用结果。 + +在整体项目中,我们将上述代码封装为 LLM.py,后续将直接从该文件中引入自定义的 LLM 类。 + +## 构建检索问答链 + +LangChain 通过提供检索问答链对象来实现对于 RAG 全流程的封装。即我们可以调用一个 LangChain 提供的 RetrievalQA 对象,通过初始化时填入已构建的数据库和自定义 LLM 作为参数,来简便地完成检索增强问答的全流程,LangChain 会自动完成基于用户提问进行检索、获取相关文档、拼接为合适的 Prompt 并交给 LLM 问答的全部流程。 + +首先我们需要将上文构建的向量数据库导入进来,我们可以直接通过 Chroma 以及上文定义的词向量模型来加载已构建的数据库: + +```python +from langchain.vectorstores import Chroma +from langchain.embeddings.huggingface import HuggingFaceEmbeddings +import os + +# 定义 Embeddings +embeddings = HuggingFaceEmbeddings(model_name="/root/autodl-tmp/embedding_model") + +# 向量数据库持久化路径 +persist_directory = 'data_base/vector_db/chroma' + +# 加载数据库 +vectordb = Chroma( + persist_directory=persist_directory, + embedding_function=embeddings +) +``` + +上述代码得到的 vectordb 对象即为我们已构建的向量数据库对象,该对象可以针对用户的 query 进行语义向量检索,得到与用户提问相关的知识片段。 + +接着,我们实例化一个基于 Yi 自定义的 LLM 对象: + +```python +from LLM import Yi_LLM +llm = Yi_LLM(mode_name_or_path = "/root/autodl-tmp/01ai/Yi-6B-Chat") +llm("你是谁") +``` + +![模型返回回答效果](images/question_to_the_Yi.png) +构建检索问答链,还需要构建一个 Prompt Template,该 Template 其实基于一个带变量的字符串,在检索之后,LangChain 会将检索到的相关文档片段填入到 Template 的变量中,从而实现带知识的 Prompt 构建。我们可以基于 LangChain 的 Template 基类来实例化这样一个 Template 对象: + +```python +from langchain.prompts import PromptTemplate + +# 我们所构造的 Prompt 模板 +template = """使用以下上下文来回答最后的问题。如果你不知道答案,就说你不知道,不要试图编造答案。尽量使答案简明扼要。总是在回答的最后说“谢谢你的提问!”。 +{context} +问题: {question} +有用的回答:""" + +# 调用 LangChain 的方法来实例化一个 Template 对象,该对象包含了 context 和 question 两个变量,在实际调用时,这两个变量会被检索到的文档片段和用户提问填充 +QA_CHAIN_PROMPT = PromptTemplate(input_variables=["context","question"],template=template) +``` + +最后,可以调用 LangChain 提供的检索问答链构造函数,基于我们的自定义 LLM、Prompt Template 和向量知识库来构建一个基于 Yi 的检索问答链: + +```python +from langchain.chains import RetrievalQA + +qa_chain = RetrievalQA.from_chain_type(llm,retriever=vectordb.as_retriever(),return_source_documents=True,chain_type_kwargs={"prompt":QA_CHAIN_PROMPT}) +``` + +得到的 qa_chain 对象即可以实现我们的核心功能,即基于 Yi 模型的专业知识库助手。我们可以对比该检索问答链和纯 LLM 的问答效果: + +```python +question = "sweettalk_django项目是什么" +result = qa_chain({"query": question}) +print("检索问答链回答 question 的结果:") +print(result["result"]) + +print("-------------------") +# 仅 LLM 回答效果 +result_2 = llm(question) +print("大模型回答 question 的结果:") +print(result_2) +``` + ![检索回答链返回结果](images/search_question_chain.png) \ No newline at end of file diff --git a/Yi/03-Yi-6B-chat WebDemo.md b/models/Yi/03-Yi-6B-chat WebDemo.md similarity index 100% rename from Yi/03-Yi-6B-chat WebDemo.md rename to models/Yi/03-Yi-6B-chat WebDemo.md diff --git a/Yi/04-Yi-6B-Chat Lora 微调.md b/models/Yi/04-Yi-6B-Chat Lora 微调.md similarity index 100% rename from Yi/04-Yi-6B-Chat Lora 微调.md rename to models/Yi/04-Yi-6B-Chat Lora 微调.md diff --git a/Yi/04-Yi-6B-chat Lora微调.py b/models/Yi/04-Yi-6B-chat Lora微调.py similarity index 100% rename from Yi/04-Yi-6B-chat Lora微调.py rename to models/Yi/04-Yi-6B-chat Lora微调.py diff --git a/Yi/images/1.png b/models/Yi/images/1.png similarity index 100% rename from Yi/images/1.png rename to models/Yi/images/1.png diff --git a/Yi/images/2.png b/models/Yi/images/2.png similarity index 100% rename from Yi/images/2.png rename to models/Yi/images/2.png diff --git a/Yi/images/3.png b/models/Yi/images/3.png similarity index 100% rename from Yi/images/3.png rename to models/Yi/images/3.png diff --git a/Yi/images/4.png b/models/Yi/images/4.png similarity index 100% rename from Yi/images/4.png rename to models/Yi/images/4.png diff --git a/Yi/images/5.png b/models/Yi/images/5.png similarity index 100% rename from Yi/images/5.png rename to models/Yi/images/5.png diff --git a/Yi/images/6.png b/models/Yi/images/6.png similarity index 100% rename from Yi/images/6.png rename to models/Yi/images/6.png diff --git a/Yi/images/Yi-Web1.png b/models/Yi/images/Yi-Web1.png similarity index 100% rename from Yi/images/Yi-Web1.png rename to models/Yi/images/Yi-Web1.png diff --git a/Yi/images/Yi-Web2.png b/models/Yi/images/Yi-Web2.png similarity index 100% rename from Yi/images/Yi-Web2.png rename to models/Yi/images/Yi-Web2.png diff --git a/Yi/images/question_to_the_Yi.png b/models/Yi/images/question_to_the_Yi.png similarity index 100% rename from Yi/images/question_to_the_Yi.png rename to models/Yi/images/question_to_the_Yi.png diff --git a/Yi/images/search_question_chain.png b/models/Yi/images/search_question_chain.png similarity index 100% rename from Yi/images/search_question_chain.png rename to models/Yi/images/search_question_chain.png diff --git a/Yuan2.0-M32/01-Yuan2.0-M32 FastApi 部署调用.md b/models/Yuan2.0-M32/01-Yuan2.0-M32 FastApi 部署调用.md similarity index 100% rename from Yuan2.0-M32/01-Yuan2.0-M32 FastApi 部署调用.md rename to models/Yuan2.0-M32/01-Yuan2.0-M32 FastApi 部署调用.md diff --git a/Yuan2.0-M32/02-Yuan2.0-M32 Langchain 接入.md b/models/Yuan2.0-M32/02-Yuan2.0-M32 Langchain 接入.md similarity index 100% rename from Yuan2.0-M32/02-Yuan2.0-M32 Langchain 接入.md rename to models/Yuan2.0-M32/02-Yuan2.0-M32 Langchain 接入.md diff --git a/Yuan2.0-M32/03-Yuan2.0-M32 WebDemo部署.md b/models/Yuan2.0-M32/03-Yuan2.0-M32 WebDemo部署.md similarity index 97% rename from Yuan2.0-M32/03-Yuan2.0-M32 WebDemo部署.md rename to models/Yuan2.0-M32/03-Yuan2.0-M32 WebDemo部署.md index 560304d..72dfe71 100644 --- a/Yuan2.0-M32/03-Yuan2.0-M32 WebDemo部署.md +++ b/models/Yuan2.0-M32/03-Yuan2.0-M32 WebDemo部署.md @@ -1,179 +1,179 @@ -# Yuan2.0-M32 WebDemo部署 - -## 环境准备 - -在 Autodl 平台中租赁一个 RTX 3090/24G 显存的显卡机器。如下图所示,镜像选择 PyTorch-->2.1.0-->3.10(ubuntu22.04)-->12.1。 - -![开启机器配置选择](images/01-1.png) - -接下来,我们打开刚刚租用服务器的 JupyterLab,如下图所示。 - -![开启JupyterLab](images/01-2.png) - -然后打开其中的终端,开始环境配置、模型下载和运行演示。 - -![开启终端](images/01-3.png) - -## 环境配置 - -Yuan2-M32-HF-INT4是由原始的Yuan2-M32-HF经过auto-gptq量化而来的模型。 - -通过模型量化,部署Yuan2-M32-HF-INT4对显存和硬盘的要求都会显著减低。 - -注:由于pip版本的auto-gptq目前还不支持Yuan2.0 M32,因此需要编译安装 - -```shell -# 升级pip -python -m pip install --upgrade pip - -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -# 拉取Yuan2.0-M32项目 -git clone https://github.com/IEIT-Yuan/Yuan2.0-M32.git - -# 进入AutoGPTQ -cd Yuan2.0-M32/3rd_party/AutoGPTQ - -# 安装autogptq -pip install --no-build-isolation -e . - -# 安装 einops modelscope streamlit -pip install einops modelscope streamlit==1.24.0 -``` - -> 考虑到部分同学配置环境可能会遇到一些问题,我们在AutoDL平台准备了Yuan2.0-M32的镜像,点击下方链接并直接创建Autodl示例即可。 -> ***https://www.codewithgpu.com/i/datawhalechina/self-llm/Yuan2.0-M32*** - - -## 模型下载 - -使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 - -这里可以先进入autodl平台,初始化机器对应区域的的文件存储,文件存储路径为'/root/autodl-fs'。 -该存储中的文件不会随着机器的关闭而丢失,这样可以避免模型二次下载。 - -![autodl-fs](images/autodl-fs.png) - -然后运行下面代码,执行模型下载。 - -```python -from modelscope import snapshot_download -model_dir = snapshot_download('YuanLLM/Yuan2-M32-HF-INT4', cache_dir='/root/autodl-fs') -``` - -## 模型合并 - -下载后的模型为多个文件,需要将其进行合并。 - -```shell -cat /root/autodl-fs/YuanLLM/Yuan2-M32-HF-INT4/gptq_model-4bit-128g.safetensors* > /root/autodl-fs/YuanLLM/Yuan2-M32-HF-INT4/gptq_model-4bit-128g.safetensors -``` - -## 代码准备 - -在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 - -chatBot.py代码如下 - -```python -# 导入所需的库 -from auto_gptq import AutoGPTQForCausalLM -from transformers import LlamaTokenizer -import torch -import streamlit as st - -# 在侧边栏中创建一个标题和一个链接 -with st.sidebar: - st.markdown("## Yuan2.0-M32 LLM") - "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" - # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 - max_length = st.slider("max_length", 0, 1024, 512, step=1) - -# 创建一个标题和一个副标题 -st.title("💬 Yuan2.0-M32 Chatbot") -st.caption("🚀 A streamlit chatbot powered by Self-LLM") - -# 定义模型路径 -path = '/root/autodl-fs/YuanLLM/Yuan2-M32-HF-INT4' - -# 定义一个函数,用于获取模型和tokenizer -@st.cache_resource -def get_model(): - print("Creat tokenizer...") - tokenizer = LlamaTokenizer.from_pretrained(path, add_eos_token=False, add_bos_token=False, eos_token='') - tokenizer.add_tokens(['', '', '', '', '', '', '','','','','','','','',''], special_tokens=True) - - print("Creat model...") - model = AutoGPTQForCausalLM.from_quantized(path, trust_remote_code=True).cuda() - - return tokenizer, model - -# 加载model和tokenizer -tokenizer, model = get_model() - -# 如果session_state中没有"messages",则创建一个包含默认消息的列表 -if "messages" not in st.session_state: - st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] - -# 遍历session_state中的所有消息,并显示在聊天界面上 -for msg in st.session_state.messages: - st.chat_message(msg["role"]).write(msg["content"]) - -# 如果用户在聊天输入框中输入了内容,则执行以下操作 -if prompt := st.chat_input(): - # 将用户的输入添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "user", "content": prompt}) - - # 在聊天界面上显示用户的输入 - st.chat_message("user").write(prompt) - - # 调用模型 - input_str = "".join(msg["content"] for msg in st.session_state.messages) + "" - inputs = tokenizer(input_str, return_tensors="pt").to(model.device) - outputs = model.generate(**inputs, do_sample=False, max_new_tokens=256) - output = tokenizer.decode(outputs[0]) - response = output.split("")[-1].replace("", '') - - # 将模型的输出添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "assistant", "content": response}) - - # 在聊天界面上显示模型的输出 - st.chat_message("assistant").write(response) - - # print(st.session_state) -``` - -# 配置vscode ssh - -复制机器ssh登录指令 - -![](images/03-0.png) - -粘贴到本地电脑的.ssh/config,并修改成如下格式 - -![](images/03-1.png) - -然后连接到此ssh,选择linx - -![](images/03-2.png) - -复制密码并输入,按下回车即可登录到机器 - -## 运行demo - -在终端中运行以下命令,启动streamlit服务 - -```shell -streamlit run chatBot.py --server.address 127.0.0.1 --server.port 6006 -``` - -![](images/03-3.png) - - -点击在浏览器中打开,即可看到聊天界面。 - -运行效果如下: - -![](images/03-4.png) - +# Yuan2.0-M32 WebDemo部署 + +## 环境准备 + +在 Autodl 平台中租赁一个 RTX 3090/24G 显存的显卡机器。如下图所示,镜像选择 PyTorch-->2.1.0-->3.10(ubuntu22.04)-->12.1。 + +![开启机器配置选择](images/01-1.png) + +接下来,我们打开刚刚租用服务器的 JupyterLab,如下图所示。 + +![开启JupyterLab](images/01-2.png) + +然后打开其中的终端,开始环境配置、模型下载和运行演示。 + +![开启终端](images/01-3.png) + +## 环境配置 + +Yuan2-M32-HF-INT4是由原始的Yuan2-M32-HF经过auto-gptq量化而来的模型。 + +通过模型量化,部署Yuan2-M32-HF-INT4对显存和硬盘的要求都会显著减低。 + +注:由于pip版本的auto-gptq目前还不支持Yuan2.0 M32,因此需要编译安装 + +```shell +# 升级pip +python -m pip install --upgrade pip + +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +# 拉取Yuan2.0-M32项目 +git clone https://github.com/IEIT-Yuan/Yuan2.0-M32.git + +# 进入AutoGPTQ +cd Yuan2.0-M32/3rd_party/AutoGPTQ + +# 安装autogptq +pip install --no-build-isolation -e . + +# 安装 einops modelscope streamlit +pip install einops modelscope streamlit==1.24.0 +``` + +> 考虑到部分同学配置环境可能会遇到一些问题,我们在AutoDL平台准备了Yuan2.0-M32的镜像,点击下方链接并直接创建Autodl示例即可。 +> ***https://www.codewithgpu.com/i/datawhalechina/self-llm/Yuan2.0-M32*** + + +## 模型下载 + +使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 + +这里可以先进入autodl平台,初始化机器对应区域的的文件存储,文件存储路径为'/root/autodl-fs'。 +该存储中的文件不会随着机器的关闭而丢失,这样可以避免模型二次下载。 + +![autodl-fs](images/autodl-fs.png) + +然后运行下面代码,执行模型下载。 + +```python +from modelscope import snapshot_download +model_dir = snapshot_download('YuanLLM/Yuan2-M32-HF-INT4', cache_dir='/root/autodl-fs') +``` + +## 模型合并 + +下载后的模型为多个文件,需要将其进行合并。 + +```shell +cat /root/autodl-fs/YuanLLM/Yuan2-M32-HF-INT4/gptq_model-4bit-128g.safetensors* > /root/autodl-fs/YuanLLM/Yuan2-M32-HF-INT4/gptq_model-4bit-128g.safetensors +``` + +## 代码准备 + +在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 + +chatBot.py代码如下 + +```python +# 导入所需的库 +from auto_gptq import AutoGPTQForCausalLM +from transformers import LlamaTokenizer +import torch +import streamlit as st + +# 在侧边栏中创建一个标题和一个链接 +with st.sidebar: + st.markdown("## Yuan2.0-M32 LLM") + "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" + # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 + max_length = st.slider("max_length", 0, 1024, 512, step=1) + +# 创建一个标题和一个副标题 +st.title("💬 Yuan2.0-M32 Chatbot") +st.caption("🚀 A streamlit chatbot powered by Self-LLM") + +# 定义模型路径 +path = '/root/autodl-fs/YuanLLM/Yuan2-M32-HF-INT4' + +# 定义一个函数,用于获取模型和tokenizer +@st.cache_resource +def get_model(): + print("Creat tokenizer...") + tokenizer = LlamaTokenizer.from_pretrained(path, add_eos_token=False, add_bos_token=False, eos_token='') + tokenizer.add_tokens(['', '', '', '', '', '', '','','','','','','','',''], special_tokens=True) + + print("Creat model...") + model = AutoGPTQForCausalLM.from_quantized(path, trust_remote_code=True).cuda() + + return tokenizer, model + +# 加载model和tokenizer +tokenizer, model = get_model() + +# 如果session_state中没有"messages",则创建一个包含默认消息的列表 +if "messages" not in st.session_state: + st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] + +# 遍历session_state中的所有消息,并显示在聊天界面上 +for msg in st.session_state.messages: + st.chat_message(msg["role"]).write(msg["content"]) + +# 如果用户在聊天输入框中输入了内容,则执行以下操作 +if prompt := st.chat_input(): + # 将用户的输入添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "user", "content": prompt}) + + # 在聊天界面上显示用户的输入 + st.chat_message("user").write(prompt) + + # 调用模型 + input_str = "".join(msg["content"] for msg in st.session_state.messages) + "" + inputs = tokenizer(input_str, return_tensors="pt").to(model.device) + outputs = model.generate(**inputs, do_sample=False, max_new_tokens=256) + output = tokenizer.decode(outputs[0]) + response = output.split("")[-1].replace("", '') + + # 将模型的输出添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "assistant", "content": response}) + + # 在聊天界面上显示模型的输出 + st.chat_message("assistant").write(response) + + # print(st.session_state) +``` + +# 配置vscode ssh + +复制机器ssh登录指令 + +![](images/03-0.png) + +粘贴到本地电脑的.ssh/config,并修改成如下格式 + +![](images/03-1.png) + +然后连接到此ssh,选择linx + +![](images/03-2.png) + +复制密码并输入,按下回车即可登录到机器 + +## 运行demo + +在终端中运行以下命令,启动streamlit服务 + +```shell +streamlit run chatBot.py --server.address 127.0.0.1 --server.port 6006 +``` + +![](images/03-3.png) + + +点击在浏览器中打开,即可看到聊天界面。 + +运行效果如下: + +![](images/03-4.png) + diff --git a/Yuan2.0-M32/README.md b/models/Yuan2.0-M32/README.md similarity index 100% rename from Yuan2.0-M32/README.md rename to models/Yuan2.0-M32/README.md diff --git a/Yuan2.0-M32/images/01-1.png b/models/Yuan2.0-M32/images/01-1.png similarity index 100% rename from Yuan2.0-M32/images/01-1.png rename to models/Yuan2.0-M32/images/01-1.png diff --git a/Yuan2.0-M32/images/01-2.png b/models/Yuan2.0-M32/images/01-2.png similarity index 100% rename from Yuan2.0-M32/images/01-2.png rename to models/Yuan2.0-M32/images/01-2.png diff --git a/Yuan2.0-M32/images/01-3.png b/models/Yuan2.0-M32/images/01-3.png similarity index 100% rename from Yuan2.0-M32/images/01-3.png rename to models/Yuan2.0-M32/images/01-3.png diff --git a/Yuan2.0-M32/images/01-4-0.png b/models/Yuan2.0-M32/images/01-4-0.png similarity index 100% rename from Yuan2.0-M32/images/01-4-0.png rename to models/Yuan2.0-M32/images/01-4-0.png diff --git a/Yuan2.0-M32/images/01-4-1.png b/models/Yuan2.0-M32/images/01-4-1.png similarity index 100% rename from Yuan2.0-M32/images/01-4-1.png rename to models/Yuan2.0-M32/images/01-4-1.png diff --git a/Yuan2.0-M32/images/01-5.png b/models/Yuan2.0-M32/images/01-5.png similarity index 100% rename from Yuan2.0-M32/images/01-5.png rename to models/Yuan2.0-M32/images/01-5.png diff --git a/Yuan2.0-M32/images/01-6.png b/models/Yuan2.0-M32/images/01-6.png similarity index 100% rename from Yuan2.0-M32/images/01-6.png rename to models/Yuan2.0-M32/images/01-6.png diff --git a/Yuan2.0-M32/images/01-7.png b/models/Yuan2.0-M32/images/01-7.png similarity index 100% rename from Yuan2.0-M32/images/01-7.png rename to models/Yuan2.0-M32/images/01-7.png diff --git a/Yuan2.0-M32/images/02-0.png b/models/Yuan2.0-M32/images/02-0.png similarity index 100% rename from Yuan2.0-M32/images/02-0.png rename to models/Yuan2.0-M32/images/02-0.png diff --git a/Yuan2.0-M32/images/03-0.png b/models/Yuan2.0-M32/images/03-0.png similarity index 100% rename from Yuan2.0-M32/images/03-0.png rename to models/Yuan2.0-M32/images/03-0.png diff --git a/Yuan2.0-M32/images/03-1.png b/models/Yuan2.0-M32/images/03-1.png similarity index 100% rename from Yuan2.0-M32/images/03-1.png rename to models/Yuan2.0-M32/images/03-1.png diff --git a/Yuan2.0-M32/images/03-2.png b/models/Yuan2.0-M32/images/03-2.png similarity index 100% rename from Yuan2.0-M32/images/03-2.png rename to models/Yuan2.0-M32/images/03-2.png diff --git a/Yuan2.0-M32/images/03-3.png b/models/Yuan2.0-M32/images/03-3.png similarity index 100% rename from Yuan2.0-M32/images/03-3.png rename to models/Yuan2.0-M32/images/03-3.png diff --git a/Yuan2.0-M32/images/03-4.png b/models/Yuan2.0-M32/images/03-4.png similarity index 100% rename from Yuan2.0-M32/images/03-4.png rename to models/Yuan2.0-M32/images/03-4.png diff --git a/Yuan2.0-M32/images/autodl-fs.png b/models/Yuan2.0-M32/images/autodl-fs.png similarity index 100% rename from Yuan2.0-M32/images/autodl-fs.png rename to models/Yuan2.0-M32/images/autodl-fs.png diff --git a/Yuan2.0-M32/images/gpu.png b/models/Yuan2.0-M32/images/gpu.png similarity index 100% rename from Yuan2.0-M32/images/gpu.png rename to models/Yuan2.0-M32/images/gpu.png diff --git a/Yuan2.0-M32/images/yuan2.0-m32-0.jpg b/models/Yuan2.0-M32/images/yuan2.0-m32-0.jpg similarity index 100% rename from Yuan2.0-M32/images/yuan2.0-m32-0.jpg rename to models/Yuan2.0-M32/images/yuan2.0-m32-0.jpg diff --git a/Yuan2.0-M32/images/yuan2.0-m32-1.jpg b/models/Yuan2.0-M32/images/yuan2.0-m32-1.jpg similarity index 100% rename from Yuan2.0-M32/images/yuan2.0-m32-1.jpg rename to models/Yuan2.0-M32/images/yuan2.0-m32-1.jpg diff --git a/Yuan2.0/01-Yuan2.0-2B FastApi 部署调用.md b/models/Yuan2.0/01-Yuan2.0-2B FastApi 部署调用.md similarity index 100% rename from Yuan2.0/01-Yuan2.0-2B FastApi 部署调用.md rename to models/Yuan2.0/01-Yuan2.0-2B FastApi 部署调用.md diff --git a/Yuan2.0/02-Yuan2.0-2B Langchain 接入.md b/models/Yuan2.0/02-Yuan2.0-2B Langchain 接入.md similarity index 100% rename from Yuan2.0/02-Yuan2.0-2B Langchain 接入.md rename to models/Yuan2.0/02-Yuan2.0-2B Langchain 接入.md diff --git a/Yuan2.0/03-Yuan2.0-2B WebDemo部署.md b/models/Yuan2.0/03-Yuan2.0-2B WebDemo部署.md similarity index 97% rename from Yuan2.0/03-Yuan2.0-2B WebDemo部署.md rename to models/Yuan2.0/03-Yuan2.0-2B WebDemo部署.md index 2adfeec..637af63 100644 --- a/Yuan2.0/03-Yuan2.0-2B WebDemo部署.md +++ b/models/Yuan2.0/03-Yuan2.0-2B WebDemo部署.md @@ -1,157 +1,157 @@ -# Yuan2.0-2B WebDemo部署 - -## 环境准备 - -在 Autodl 平台中租赁一个 RTX 3090/24G 显存的显卡机器。如下图所示,镜像选择 PyTorch-->2.1.0-->3.10(ubuntu22.04)-->12.1。 - -![开启机器配置选择](images/01-1.png) - -接下来,我们打开刚刚租用服务器的 JupyterLab,如下图所示。 - -![开启JupyterLab](images/01-2.png) - -然后打开其中的终端,开始环境配置、模型下载和运行演示。 - -![开启终端](images/01-3.png) - -## 环境配置 - -pip 换源加速下载并安装依赖包 - -```shell -# 升级pip -python -m pip install --upgrade pip - -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -# 安装 einops modelscope streamlit -pip install einops modelscope streamlit==1.24.0 -``` - -> 考虑到部分同学配置环境可能会遇到一些问题,我们在AutoDL平台准备了Yuan2.0的镜像,点击下方链接并直接创建Autodl示例即可。 -> ***https://www.codewithgpu.com/i/datawhalechina/self-llm/Yuan2.0*** - - -## 模型下载 - -使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 - -这里可以先进入autodl平台,初始化机器对应区域的的文件存储,文件存储路径为'/root/autodl-fs'。 -该存储中的文件不会随着机器的关闭而丢失,这样可以避免模型二次下载。 - -![autodl-fs](images/autodl-fs.png) - -然后运行下面代码,执行模型下载。模型大小为 4.5GB,下载大概需要 5 分钟。 - -```python -from modelscope import snapshot_download -model_dir = snapshot_download('YuanLLM/Yuan2-2B-Mars-hf', cache_dir='/root/autodl-fs') -``` - -## 代码准备 - -在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 - -chatBot.py代码如下 - -```python -# 导入所需的库 -from transformers import LlamaTokenizer, AutoModelForCausalLM -import torch -import streamlit as st - -# 在侧边栏中创建一个标题和一个链接 -with st.sidebar: - st.markdown("## Yuan2.0 LLM") - "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" - # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 - max_length = st.slider("max_length", 0, 1024, 512, step=1) - -# 创建一个标题和一个副标题 -st.title("💬 Yuan2.0 Chatbot") -st.caption("🚀 A streamlit chatbot powered by Self-LLM") - -# 定义模型路径 -path = '/root/autodl-fs/YuanLLM/Yuan2-2B-Mars-hf' - -# 定义一个函数,用于获取模型和tokenizer -@st.cache_resource -def get_model(): - print("Creat tokenizer...") - tokenizer = LlamaTokenizer.from_pretrained(path, add_eos_token=False, add_bos_token=False, eos_token='') - tokenizer.add_tokens(['', '', '', '', '', '', '','','','','','','','',''], special_tokens=True) - - print("Creat model...") - model = AutoModelForCausalLM.from_pretrained(path, torch_dtype=torch.bfloat16, trust_remote_code=True).cuda() - - return tokenizer, model - -# 加载model和tokenizer -tokenizer, model = get_model() - -# 如果session_state中没有"messages",则创建一个包含默认消息的列表 -if "messages" not in st.session_state: - st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] - -# 遍历session_state中的所有消息,并显示在聊天界面上 -for msg in st.session_state.messages: - st.chat_message(msg["role"]).write(msg["content"]) - -# 如果用户在聊天输入框中输入了内容,则执行以下操作 -if prompt := st.chat_input(): - # 将用户的输入添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "user", "content": prompt}) - - # 在聊天界面上显示用户的输入 - st.chat_message("user").write(prompt) - - # 调用模型 - input_str = "".join(msg["content"] for msg in st.session_state.messages) + "" - inputs = tokenizer(input_str, return_tensors="pt")["input_ids"].cuda() - outputs = model.generate(inputs,do_sample=False,max_length=4000) - output = tokenizer.decode(outputs[0]) - response = output.split("")[-1].replace("", '') - - # 将模型的输出添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "assistant", "content": response}) - - # 在聊天界面上显示模型的输出 - st.chat_message("assistant").write(response) - - # print(st.session_state) -``` - -# 配置vscode ssh - -复制机器ssh登录指令 - -![](images/03-0.png) - -粘贴到本地电脑的.ssh/config,并修改成如下格式 - -![](images/03-1.png) - -然后连接到此ssh,选择linx - -![](images/03-2.png) - -复制密码并输入,按下回车即可登录到机器 - -## 运行demo - -在终端中运行以下命令,启动streamlit服务 - -```shell -streamlit run chatBot.py --server.address 127.0.0.1 --server.port 6006 -``` - -![](images/03-3.png) - - -点击在浏览器中打开,即可看到聊天界面。 - -运行效果如下: - -![](images/03-4.png) - +# Yuan2.0-2B WebDemo部署 + +## 环境准备 + +在 Autodl 平台中租赁一个 RTX 3090/24G 显存的显卡机器。如下图所示,镜像选择 PyTorch-->2.1.0-->3.10(ubuntu22.04)-->12.1。 + +![开启机器配置选择](images/01-1.png) + +接下来,我们打开刚刚租用服务器的 JupyterLab,如下图所示。 + +![开启JupyterLab](images/01-2.png) + +然后打开其中的终端,开始环境配置、模型下载和运行演示。 + +![开启终端](images/01-3.png) + +## 环境配置 + +pip 换源加速下载并安装依赖包 + +```shell +# 升级pip +python -m pip install --upgrade pip + +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +# 安装 einops modelscope streamlit +pip install einops modelscope streamlit==1.24.0 +``` + +> 考虑到部分同学配置环境可能会遇到一些问题,我们在AutoDL平台准备了Yuan2.0的镜像,点击下方链接并直接创建Autodl示例即可。 +> ***https://www.codewithgpu.com/i/datawhalechina/self-llm/Yuan2.0*** + + +## 模型下载 + +使用 modelscope 中的 snapshot_download 函数下载模型,第一个参数为模型名称,参数 cache_dir 为模型的下载路径。 + +这里可以先进入autodl平台,初始化机器对应区域的的文件存储,文件存储路径为'/root/autodl-fs'。 +该存储中的文件不会随着机器的关闭而丢失,这样可以避免模型二次下载。 + +![autodl-fs](images/autodl-fs.png) + +然后运行下面代码,执行模型下载。模型大小为 4.5GB,下载大概需要 5 分钟。 + +```python +from modelscope import snapshot_download +model_dir = snapshot_download('YuanLLM/Yuan2-2B-Mars-hf', cache_dir='/root/autodl-fs') +``` + +## 代码准备 + +在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 + +chatBot.py代码如下 + +```python +# 导入所需的库 +from transformers import LlamaTokenizer, AutoModelForCausalLM +import torch +import streamlit as st + +# 在侧边栏中创建一个标题和一个链接 +with st.sidebar: + st.markdown("## Yuan2.0 LLM") + "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" + # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 + max_length = st.slider("max_length", 0, 1024, 512, step=1) + +# 创建一个标题和一个副标题 +st.title("💬 Yuan2.0 Chatbot") +st.caption("🚀 A streamlit chatbot powered by Self-LLM") + +# 定义模型路径 +path = '/root/autodl-fs/YuanLLM/Yuan2-2B-Mars-hf' + +# 定义一个函数,用于获取模型和tokenizer +@st.cache_resource +def get_model(): + print("Creat tokenizer...") + tokenizer = LlamaTokenizer.from_pretrained(path, add_eos_token=False, add_bos_token=False, eos_token='') + tokenizer.add_tokens(['', '', '', '', '', '', '','','','','','','','',''], special_tokens=True) + + print("Creat model...") + model = AutoModelForCausalLM.from_pretrained(path, torch_dtype=torch.bfloat16, trust_remote_code=True).cuda() + + return tokenizer, model + +# 加载model和tokenizer +tokenizer, model = get_model() + +# 如果session_state中没有"messages",则创建一个包含默认消息的列表 +if "messages" not in st.session_state: + st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] + +# 遍历session_state中的所有消息,并显示在聊天界面上 +for msg in st.session_state.messages: + st.chat_message(msg["role"]).write(msg["content"]) + +# 如果用户在聊天输入框中输入了内容,则执行以下操作 +if prompt := st.chat_input(): + # 将用户的输入添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "user", "content": prompt}) + + # 在聊天界面上显示用户的输入 + st.chat_message("user").write(prompt) + + # 调用模型 + input_str = "".join(msg["content"] for msg in st.session_state.messages) + "" + inputs = tokenizer(input_str, return_tensors="pt")["input_ids"].cuda() + outputs = model.generate(inputs,do_sample=False,max_length=4000) + output = tokenizer.decode(outputs[0]) + response = output.split("")[-1].replace("", '') + + # 将模型的输出添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "assistant", "content": response}) + + # 在聊天界面上显示模型的输出 + st.chat_message("assistant").write(response) + + # print(st.session_state) +``` + +# 配置vscode ssh + +复制机器ssh登录指令 + +![](images/03-0.png) + +粘贴到本地电脑的.ssh/config,并修改成如下格式 + +![](images/03-1.png) + +然后连接到此ssh,选择linx + +![](images/03-2.png) + +复制密码并输入,按下回车即可登录到机器 + +## 运行demo + +在终端中运行以下命令,启动streamlit服务 + +```shell +streamlit run chatBot.py --server.address 127.0.0.1 --server.port 6006 +``` + +![](images/03-3.png) + + +点击在浏览器中打开,即可看到聊天界面。 + +运行效果如下: + +![](images/03-4.png) + diff --git a/Yuan2.0/04-Yuan2.0-2B vLLM部署调用.ipynb b/models/Yuan2.0/04-Yuan2.0-2B vLLM部署调用.ipynb similarity index 100% rename from Yuan2.0/04-Yuan2.0-2B vLLM部署调用.ipynb rename to models/Yuan2.0/04-Yuan2.0-2B vLLM部署调用.ipynb diff --git a/Yuan2.0/04-Yuan2.0-2B vLLM部署调用.md b/models/Yuan2.0/04-Yuan2.0-2B vLLM部署调用.md similarity index 100% rename from Yuan2.0/04-Yuan2.0-2B vLLM部署调用.md rename to models/Yuan2.0/04-Yuan2.0-2B vLLM部署调用.md diff --git a/Yuan2.0/05-Yuan2.0-2B Lora-bf16.ipynb b/models/Yuan2.0/05-Yuan2.0-2B Lora-bf16.ipynb similarity index 100% rename from Yuan2.0/05-Yuan2.0-2B Lora-bf16.ipynb rename to models/Yuan2.0/05-Yuan2.0-2B Lora-bf16.ipynb diff --git a/Yuan2.0/05-Yuan2.0-2B Lora-fp16.ipynb b/models/Yuan2.0/05-Yuan2.0-2B Lora-fp16.ipynb similarity index 100% rename from Yuan2.0/05-Yuan2.0-2B Lora-fp16.ipynb rename to models/Yuan2.0/05-Yuan2.0-2B Lora-fp16.ipynb diff --git a/Yuan2.0/05-Yuan2.0-2B Lora微调.md b/models/Yuan2.0/05-Yuan2.0-2B Lora微调.md similarity index 100% rename from Yuan2.0/05-Yuan2.0-2B Lora微调.md rename to models/Yuan2.0/05-Yuan2.0-2B Lora微调.md diff --git a/Yuan2.0/README.md b/models/Yuan2.0/README.md similarity index 100% rename from Yuan2.0/README.md rename to models/Yuan2.0/README.md diff --git a/Yuan2.0/images/01-1.png b/models/Yuan2.0/images/01-1.png similarity index 100% rename from Yuan2.0/images/01-1.png rename to models/Yuan2.0/images/01-1.png diff --git a/Yuan2.0/images/01-2.png b/models/Yuan2.0/images/01-2.png similarity index 100% rename from Yuan2.0/images/01-2.png rename to models/Yuan2.0/images/01-2.png diff --git a/Yuan2.0/images/01-3.png b/models/Yuan2.0/images/01-3.png similarity index 100% rename from Yuan2.0/images/01-3.png rename to models/Yuan2.0/images/01-3.png diff --git a/Yuan2.0/images/01-4-0.png b/models/Yuan2.0/images/01-4-0.png similarity index 100% rename from Yuan2.0/images/01-4-0.png rename to models/Yuan2.0/images/01-4-0.png diff --git a/Yuan2.0/images/01-4-1.png b/models/Yuan2.0/images/01-4-1.png similarity index 100% rename from Yuan2.0/images/01-4-1.png rename to models/Yuan2.0/images/01-4-1.png diff --git a/Yuan2.0/images/01-5.png b/models/Yuan2.0/images/01-5.png similarity index 100% rename from Yuan2.0/images/01-5.png rename to models/Yuan2.0/images/01-5.png diff --git a/Yuan2.0/images/01-6.png b/models/Yuan2.0/images/01-6.png similarity index 100% rename from Yuan2.0/images/01-6.png rename to models/Yuan2.0/images/01-6.png diff --git a/Yuan2.0/images/01-7.png b/models/Yuan2.0/images/01-7.png similarity index 100% rename from Yuan2.0/images/01-7.png rename to models/Yuan2.0/images/01-7.png diff --git a/Yuan2.0/images/02-0.png b/models/Yuan2.0/images/02-0.png similarity index 100% rename from Yuan2.0/images/02-0.png rename to models/Yuan2.0/images/02-0.png diff --git a/Yuan2.0/images/03-0.png b/models/Yuan2.0/images/03-0.png similarity index 100% rename from Yuan2.0/images/03-0.png rename to models/Yuan2.0/images/03-0.png diff --git a/Yuan2.0/images/03-1.png b/models/Yuan2.0/images/03-1.png similarity index 100% rename from Yuan2.0/images/03-1.png rename to models/Yuan2.0/images/03-1.png diff --git a/Yuan2.0/images/03-2.png b/models/Yuan2.0/images/03-2.png similarity index 100% rename from Yuan2.0/images/03-2.png rename to models/Yuan2.0/images/03-2.png diff --git a/Yuan2.0/images/03-3.png b/models/Yuan2.0/images/03-3.png similarity index 100% rename from Yuan2.0/images/03-3.png rename to models/Yuan2.0/images/03-3.png diff --git a/Yuan2.0/images/03-4.png b/models/Yuan2.0/images/03-4.png similarity index 100% rename from Yuan2.0/images/03-4.png rename to models/Yuan2.0/images/03-4.png diff --git a/Yuan2.0/images/04-0.png b/models/Yuan2.0/images/04-0.png similarity index 100% rename from Yuan2.0/images/04-0.png rename to models/Yuan2.0/images/04-0.png diff --git a/Yuan2.0/images/04-1.png b/models/Yuan2.0/images/04-1.png similarity index 100% rename from Yuan2.0/images/04-1.png rename to models/Yuan2.0/images/04-1.png diff --git a/Yuan2.0/images/05-fp-0.png b/models/Yuan2.0/images/05-fp-0.png similarity index 100% rename from Yuan2.0/images/05-fp-0.png rename to models/Yuan2.0/images/05-fp-0.png diff --git a/Yuan2.0/images/05-fp-1.png b/models/Yuan2.0/images/05-fp-1.png similarity index 100% rename from Yuan2.0/images/05-fp-1.png rename to models/Yuan2.0/images/05-fp-1.png diff --git a/Yuan2.0/images/05-fp-2.png b/models/Yuan2.0/images/05-fp-2.png similarity index 100% rename from Yuan2.0/images/05-fp-2.png rename to models/Yuan2.0/images/05-fp-2.png diff --git a/Yuan2.0/images/05-fp-3.png b/models/Yuan2.0/images/05-fp-3.png similarity index 100% rename from Yuan2.0/images/05-fp-3.png rename to models/Yuan2.0/images/05-fp-3.png diff --git a/Yuan2.0/images/05-gpu-0.png b/models/Yuan2.0/images/05-gpu-0.png similarity index 100% rename from Yuan2.0/images/05-gpu-0.png rename to models/Yuan2.0/images/05-gpu-0.png diff --git a/Yuan2.0/images/05-gpu-1.png b/models/Yuan2.0/images/05-gpu-1.png similarity index 100% rename from Yuan2.0/images/05-gpu-1.png rename to models/Yuan2.0/images/05-gpu-1.png diff --git a/Yuan2.0/images/autodl-fs.png b/models/Yuan2.0/images/autodl-fs.png similarity index 100% rename from Yuan2.0/images/autodl-fs.png rename to models/Yuan2.0/images/autodl-fs.png diff --git a/Yuan2.0/images/yuan2.0-0.png b/models/Yuan2.0/images/yuan2.0-0.png similarity index 100% rename from Yuan2.0/images/yuan2.0-0.png rename to models/Yuan2.0/images/yuan2.0-0.png diff --git a/Yuan2.0/images/yuan2.0-1.jpg b/models/Yuan2.0/images/yuan2.0-1.jpg similarity index 100% rename from Yuan2.0/images/yuan2.0-1.jpg rename to models/Yuan2.0/images/yuan2.0-1.jpg diff --git a/bilibili_Index-1.9B/01-Index-1.9B-chat FastApi 部署调用.md b/models/bilibili_Index-1.9B/01-Index-1.9B-chat FastApi 部署调用.md similarity index 100% rename from bilibili_Index-1.9B/01-Index-1.9B-chat FastApi 部署调用.md rename to models/bilibili_Index-1.9B/01-Index-1.9B-chat FastApi 部署调用.md diff --git a/bilibili_Index-1.9B/02-Index-1.9B-Chat 接入 LangChain.md b/models/bilibili_Index-1.9B/02-Index-1.9B-Chat 接入 LangChain.md similarity index 100% rename from bilibili_Index-1.9B/02-Index-1.9B-Chat 接入 LangChain.md rename to models/bilibili_Index-1.9B/02-Index-1.9B-Chat 接入 LangChain.md diff --git a/bilibili_Index-1.9B/03-Index-1.9B-chat WebDemo部署.md b/models/bilibili_Index-1.9B/03-Index-1.9B-chat WebDemo部署.md similarity index 97% rename from bilibili_Index-1.9B/03-Index-1.9B-chat WebDemo部署.md rename to models/bilibili_Index-1.9B/03-Index-1.9B-chat WebDemo部署.md index 801ac6f..4b683fd 100644 --- a/bilibili_Index-1.9B/03-Index-1.9B-chat WebDemo部署.md +++ b/models/bilibili_Index-1.9B/03-Index-1.9B-chat WebDemo部署.md @@ -1,145 +1,145 @@ -# Index-1.9B-chat WebDemo部署 - -## 环境准备 - -在 [AutoDL](https://www.autodl.com/) 平台中租一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 `PyTorch`-->`2.1.0`-->`3.10(ubuntu22.04)`-->`12.1`。 - -![01-1.png](images/01-1.png) - -接下来打开刚刚租用服务器的 `JupyterLab`,并且打开其中的终端开始环境配置、模型下载和运行 `demo`。 - -pip 换源和安装依赖包。 - -```bash -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install modelscope==1.9.5 -pip install transformers==4.39.2 -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.27.0 -pip install transformers_stream_generator==0.0.4 -pip install tiktoken==0.7.0 -pip install huggingface_hub==0.23.4 -``` - -## 模型下载 - -使用 `modelscope` 中的 `snapshot_download` 函数下载模型,第一个参数为模型名称,参数 `cache_dir` 为模型的下载路径,参数`revision`为模型的版本,master代表主分支,为最新版本。 - -在 `/root/autodl-tmp` 路径下新建 `download.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 `python /root/autodl-tmp/download.py` 执行下载,模型大小为 8 GB,下载模型大概需要 5 钟。 - -```python -import torch -from modelscope import snapshot_download, AutoModel, AutoTokenizer -import os - -model_dir = snapshot_download('IndexTeam/Index-1.9B-Chat', cache_dir='/root/autodl-tmp', revision='master') -``` - -终端出现下图结果表示下载成功。 - -![](images/image01-0.png) - -## 代码准备 - -在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 - -![03-11.png](images/03-11.png) - -![03-12.png](images/03-12.png) - -chatBot.py代码如下 - -``` -# 导入所需的库 -from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig -import torch -import streamlit as st - -# 在侧边栏中创建一个标题和一个链接 -with st.sidebar: - st.markdown("## Index-1.9B-chat LLM") - "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" - # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 - max_length = st.slider("max_length", 0, 1024, 512, step=1) - -# 创建一个标题和一个副标题 -st.title("💬 Index-1.9B-chat Chatbot") -st.caption("🚀 A streamlit chatbot powered by Self-LLM") - -# 定义模型路径 -model_name_or_path = '/root/autodl-tmp/Index-1.9B-Chat' - -# 定义一个函数,用于获取模型和tokenizer -@st.cache_resource -def get_model(): - # 从预训练的模型中获取tokenizer - tokenizer = AutoTokenizer.from_pretrained(model_name_or_path, use_fast=False, trust_remote_code=True) - # 从预训练的模型中获取模型,并设置模型参数 - model = AutoModelForCausalLM.from_pretrained(model_name_or_path, torch_dtype=torch.bfloat16, device_map="auto", trust_remote_code=True) - - return tokenizer, model - -# 加载 Index-1.9B-chat 的model和tokenizer -tokenizer, model = get_model() - -# 如果session_state中没有"messages",则创建一个包含默认消息的列表 -if "messages" not in st.session_state: - st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] - -# 遍历session_state中的所有消息,并显示在聊天界面上 -for msg in st.session_state.messages: - st.chat_message(msg["role"]).write(msg["content"]) - -# 如果用户在聊天输入框中输入了内容,则执行以下操作 -if prompt := st.chat_input(): - # 将用户的输入添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "user", "content": prompt}) - # 在聊天界面上显示用户的输入 - st.chat_message("user").write(prompt) - - # 构建输入 - input_ids = tokenizer.apply_chat_template(st.session_state.messages,tokenize=False,add_generation_prompt=True) - model_inputs = tokenizer([input_ids], return_tensors="pt").to('cuda') - generated_ids = model.generate(model_inputs.input_ids, max_new_tokens=512) - generated_ids = [ - output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) - ] - response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0] - # 将模型的输出添加到session_state中的messages列表中 - st.session_state.messages.append({"role": "assistant", "content": response}) - # 在聊天界面上显示模型的输出 - st.chat_message("assistant").write(response) - # print(st.session_state) -``` - -## 运行demo - -在终端中运行以下命令,启动streamlit服务 - -``` -streamlit run /root/autodl-tmp/chatBot.py --server.address 127.0.0.1 --server.port 6006 -``` - -点击自定义服务 - -![03-13.png](images/03-13.png) - -点开linux - -![03-14.png](images/03-14.png) - -然后win+R打开powershell - -![03-15.png](images/03-15.png) - -输入ssh与密码,按下回车至这样即可 - -![03-16.png](images/03-16.png) - -在浏览器中打开链接 http://localhost:6006/ ,即可看到聊天界面。运行效果如下:![03-17.png](images/03-17.png) - +# Index-1.9B-chat WebDemo部署 + +## 环境准备 + +在 [AutoDL](https://www.autodl.com/) 平台中租一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 `PyTorch`-->`2.1.0`-->`3.10(ubuntu22.04)`-->`12.1`。 + +![01-1.png](images/01-1.png) + +接下来打开刚刚租用服务器的 `JupyterLab`,并且打开其中的终端开始环境配置、模型下载和运行 `demo`。 + +pip 换源和安装依赖包。 + +```bash +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install modelscope==1.9.5 +pip install transformers==4.39.2 +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.27.0 +pip install transformers_stream_generator==0.0.4 +pip install tiktoken==0.7.0 +pip install huggingface_hub==0.23.4 +``` + +## 模型下载 + +使用 `modelscope` 中的 `snapshot_download` 函数下载模型,第一个参数为模型名称,参数 `cache_dir` 为模型的下载路径,参数`revision`为模型的版本,master代表主分支,为最新版本。 + +在 `/root/autodl-tmp` 路径下新建 `download.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行 `python /root/autodl-tmp/download.py` 执行下载,模型大小为 8 GB,下载模型大概需要 5 钟。 + +```python +import torch +from modelscope import snapshot_download, AutoModel, AutoTokenizer +import os + +model_dir = snapshot_download('IndexTeam/Index-1.9B-Chat', cache_dir='/root/autodl-tmp', revision='master') +``` + +终端出现下图结果表示下载成功。 + +![](images/image01-0.png) + +## 代码准备 + +在`/root/autodl-tmp`路径下新建 `chatBot.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 + +![03-11.png](images/03-11.png) + +![03-12.png](images/03-12.png) + +chatBot.py代码如下 + +``` +# 导入所需的库 +from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig +import torch +import streamlit as st + +# 在侧边栏中创建一个标题和一个链接 +with st.sidebar: + st.markdown("## Index-1.9B-chat LLM") + "[开源大模型食用指南 self-llm](https://github.com/datawhalechina/self-llm.git)" + # 创建一个滑块,用于选择最大长度,范围在0到1024之间,默认值为512 + max_length = st.slider("max_length", 0, 1024, 512, step=1) + +# 创建一个标题和一个副标题 +st.title("💬 Index-1.9B-chat Chatbot") +st.caption("🚀 A streamlit chatbot powered by Self-LLM") + +# 定义模型路径 +model_name_or_path = '/root/autodl-tmp/Index-1.9B-Chat' + +# 定义一个函数,用于获取模型和tokenizer +@st.cache_resource +def get_model(): + # 从预训练的模型中获取tokenizer + tokenizer = AutoTokenizer.from_pretrained(model_name_or_path, use_fast=False, trust_remote_code=True) + # 从预训练的模型中获取模型,并设置模型参数 + model = AutoModelForCausalLM.from_pretrained(model_name_or_path, torch_dtype=torch.bfloat16, device_map="auto", trust_remote_code=True) + + return tokenizer, model + +# 加载 Index-1.9B-chat 的model和tokenizer +tokenizer, model = get_model() + +# 如果session_state中没有"messages",则创建一个包含默认消息的列表 +if "messages" not in st.session_state: + st.session_state["messages"] = [{"role": "assistant", "content": "有什么可以帮您的?"}] + +# 遍历session_state中的所有消息,并显示在聊天界面上 +for msg in st.session_state.messages: + st.chat_message(msg["role"]).write(msg["content"]) + +# 如果用户在聊天输入框中输入了内容,则执行以下操作 +if prompt := st.chat_input(): + # 将用户的输入添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "user", "content": prompt}) + # 在聊天界面上显示用户的输入 + st.chat_message("user").write(prompt) + + # 构建输入 + input_ids = tokenizer.apply_chat_template(st.session_state.messages,tokenize=False,add_generation_prompt=True) + model_inputs = tokenizer([input_ids], return_tensors="pt").to('cuda') + generated_ids = model.generate(model_inputs.input_ids, max_new_tokens=512) + generated_ids = [ + output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids) + ] + response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0] + # 将模型的输出添加到session_state中的messages列表中 + st.session_state.messages.append({"role": "assistant", "content": response}) + # 在聊天界面上显示模型的输出 + st.chat_message("assistant").write(response) + # print(st.session_state) +``` + +## 运行demo + +在终端中运行以下命令,启动streamlit服务 + +``` +streamlit run /root/autodl-tmp/chatBot.py --server.address 127.0.0.1 --server.port 6006 +``` + +点击自定义服务 + +![03-13.png](images/03-13.png) + +点开linux + +![03-14.png](images/03-14.png) + +然后win+R打开powershell + +![03-15.png](images/03-15.png) + +输入ssh与密码,按下回车至这样即可 + +![03-16.png](images/03-16.png) + +在浏览器中打开链接 http://localhost:6006/ ,即可看到聊天界面。运行效果如下:![03-17.png](images/03-17.png) + diff --git a/bilibili_Index-1.9B/04-Index-1.9B-Chat Lora 微调.md b/models/bilibili_Index-1.9B/04-Index-1.9B-Chat Lora 微调.md similarity index 100% rename from bilibili_Index-1.9B/04-Index-1.9B-Chat Lora 微调.md rename to models/bilibili_Index-1.9B/04-Index-1.9B-Chat Lora 微调.md diff --git a/bilibili_Index-1.9B/04-Index-1.9B-Chat Lora.ipynb b/models/bilibili_Index-1.9B/04-Index-1.9B-Chat Lora.ipynb similarity index 100% rename from bilibili_Index-1.9B/04-Index-1.9B-Chat Lora.ipynb rename to models/bilibili_Index-1.9B/04-Index-1.9B-Chat Lora.ipynb diff --git a/bilibili_Index-1.9B/images/01-1.png b/models/bilibili_Index-1.9B/images/01-1.png similarity index 100% rename from bilibili_Index-1.9B/images/01-1.png rename to models/bilibili_Index-1.9B/images/01-1.png diff --git a/bilibili_Index-1.9B/images/03-11.png b/models/bilibili_Index-1.9B/images/03-11.png similarity index 100% rename from bilibili_Index-1.9B/images/03-11.png rename to models/bilibili_Index-1.9B/images/03-11.png diff --git a/bilibili_Index-1.9B/images/03-12.png b/models/bilibili_Index-1.9B/images/03-12.png similarity index 100% rename from bilibili_Index-1.9B/images/03-12.png rename to models/bilibili_Index-1.9B/images/03-12.png diff --git a/bilibili_Index-1.9B/images/03-13.png b/models/bilibili_Index-1.9B/images/03-13.png similarity index 100% rename from bilibili_Index-1.9B/images/03-13.png rename to models/bilibili_Index-1.9B/images/03-13.png diff --git a/bilibili_Index-1.9B/images/03-14.png b/models/bilibili_Index-1.9B/images/03-14.png similarity index 100% rename from bilibili_Index-1.9B/images/03-14.png rename to models/bilibili_Index-1.9B/images/03-14.png diff --git a/bilibili_Index-1.9B/images/03-15.png b/models/bilibili_Index-1.9B/images/03-15.png similarity index 100% rename from bilibili_Index-1.9B/images/03-15.png rename to models/bilibili_Index-1.9B/images/03-15.png diff --git a/bilibili_Index-1.9B/images/03-16.png b/models/bilibili_Index-1.9B/images/03-16.png similarity index 100% rename from bilibili_Index-1.9B/images/03-16.png rename to models/bilibili_Index-1.9B/images/03-16.png diff --git a/bilibili_Index-1.9B/images/03-17.png b/models/bilibili_Index-1.9B/images/03-17.png similarity index 100% rename from bilibili_Index-1.9B/images/03-17.png rename to models/bilibili_Index-1.9B/images/03-17.png diff --git a/bilibili_Index-1.9B/images/fig4-1.png b/models/bilibili_Index-1.9B/images/fig4-1.png similarity index 100% rename from bilibili_Index-1.9B/images/fig4-1.png rename to models/bilibili_Index-1.9B/images/fig4-1.png diff --git a/bilibili_Index-1.9B/images/fig4-2.png b/models/bilibili_Index-1.9B/images/fig4-2.png similarity index 100% rename from bilibili_Index-1.9B/images/fig4-2.png rename to models/bilibili_Index-1.9B/images/fig4-2.png diff --git a/bilibili_Index-1.9B/images/fig4-3.png b/models/bilibili_Index-1.9B/images/fig4-3.png similarity index 100% rename from bilibili_Index-1.9B/images/fig4-3.png rename to models/bilibili_Index-1.9B/images/fig4-3.png diff --git a/bilibili_Index-1.9B/images/fig4-4.png b/models/bilibili_Index-1.9B/images/fig4-4.png similarity index 100% rename from bilibili_Index-1.9B/images/fig4-4.png rename to models/bilibili_Index-1.9B/images/fig4-4.png diff --git a/bilibili_Index-1.9B/images/image01-0.png b/models/bilibili_Index-1.9B/images/image01-0.png similarity index 100% rename from bilibili_Index-1.9B/images/image01-0.png rename to models/bilibili_Index-1.9B/images/image01-0.png diff --git a/bilibili_Index-1.9B/images/image01-1.png b/models/bilibili_Index-1.9B/images/image01-1.png similarity index 100% rename from bilibili_Index-1.9B/images/image01-1.png rename to models/bilibili_Index-1.9B/images/image01-1.png diff --git a/bilibili_Index-1.9B/images/image01-2.png b/models/bilibili_Index-1.9B/images/image01-2.png similarity index 100% rename from bilibili_Index-1.9B/images/image01-2.png rename to models/bilibili_Index-1.9B/images/image01-2.png diff --git a/bilibili_Index-1.9B/images/image01-3.png b/models/bilibili_Index-1.9B/images/image01-3.png similarity index 100% rename from bilibili_Index-1.9B/images/image01-3.png rename to models/bilibili_Index-1.9B/images/image01-3.png diff --git a/bilibili_Index-1.9B/images/image01-4.png b/models/bilibili_Index-1.9B/images/image01-4.png similarity index 100% rename from bilibili_Index-1.9B/images/image01-4.png rename to models/bilibili_Index-1.9B/images/image01-4.png diff --git a/bilibili_Index-1.9B/images/image02-1.png b/models/bilibili_Index-1.9B/images/image02-1.png similarity index 100% rename from bilibili_Index-1.9B/images/image02-1.png rename to models/bilibili_Index-1.9B/images/image02-1.png diff --git a/bilibili_Index-1.9B/images/image02-2.png b/models/bilibili_Index-1.9B/images/image02-2.png similarity index 100% rename from bilibili_Index-1.9B/images/image02-2.png rename to models/bilibili_Index-1.9B/images/image02-2.png diff --git a/phi-3/01-Phi-3-mini-4k-instruct FastApi 部署调用.md b/models/phi-3/01-Phi-3-mini-4k-instruct FastApi 部署调用.md similarity index 97% rename from phi-3/01-Phi-3-mini-4k-instruct FastApi 部署调用.md rename to models/phi-3/01-Phi-3-mini-4k-instruct FastApi 部署调用.md index 8e5e2b8..844a7fe 100644 --- a/phi-3/01-Phi-3-mini-4k-instruct FastApi 部署调用.md +++ b/models/phi-3/01-Phi-3-mini-4k-instruct FastApi 部署调用.md @@ -1,170 +1,170 @@ -# Phi-3-mini-4k-instruct FastApi 部署调用 - -## 环境准备 - -在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 。 - -接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 - -![机器配置选择](../InternLM2/images/1.png) - -### 创建工作目录 - -创建本次phi3实践的工作目录`/root/autodl-tmp/phi3` - -```bash -# 创建工作目录 -mkdir -p /root/autodl-tmp/phi3 -``` - -### 安装依赖 - -```bash -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install fastapi==0.104.1 -pip install uvicorn==0.24.0.post1 -pip install requests==2.25.1 -pip install modelscope==1.9.5 -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.24.1 -``` - -由于phi3要求的transformers的版本为`4.41.0.dev0版本`。 - -各位可以先通过下面命令查看你的Transformers包的版本 - -```bash -pip list |grep transformers -``` - -如果版本不对,可以通过下面命令升级 - -```bash -# phi3升级transformers为4.41.0.dev0版本 -pip uninstall -y transformers && pip install git+https://github.com/huggingface/transformers -``` - - - -## 模型下载 - -使用 modelscope 中的`napshot_download`函数下载模型,第一个参数为模型名称,参数`cache_dir`为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建`download.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行`python /root/autodl-tmp/download.py`执行下载,模型大小为 8 GB,下载模型大概需要 10~15 分钟 - -```python -#模型下载 -from modelscope import snapshot_download -model_dir = snapshot_download('LLM-Research/Phi-3-mini-4k-instruct', cache_dir='/root/autodl-tmp/phi3', revision='master') -``` - -## 代码准备 - -在`/root/autodl-tmp`路径下新建`api.py`文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 - -```python -from fastapi import FastAPI, Request -from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig -import uvicorn -import json -import datetime -import torch - -# 设置设备参数 -DEVICE = "cuda" # 使用CUDA -DEVICE_ID = "0" # CUDA设备ID,如果未设置则为空 -CUDA_DEVICE = f"{DEVICE}:{DEVICE_ID}" if DEVICE_ID else DEVICE # 组合CUDA设备信息 - -# 清理GPU内存函数 -def torch_gc(): - if torch.cuda.is_available(): # 检查是否可用CUDA - with torch.cuda.device(CUDA_DEVICE): # 指定CUDA设备 - torch.cuda.empty_cache() # 清空CUDA缓存 - torch.cuda.ipc_collect() # 收集CUDA内存碎片 - -# 创建FastAPI应用 -app = FastAPI() - -# 处理POST请求的端点 -@app.post("/") -async def create_item(request: Request): - global model, tokenizer # 声明全局变量以便在函数内部使用模型和分词器 - json_post_raw = await request.json() # 获取POST请求的JSON数据 - json_post = json.dumps(json_post_raw) # 将JSON数据转换为字符串 - json_post_list = json.loads(json_post) # 将字符串转换为Python对象 - prompt = json_post_list.get('prompt') # 获取请求中的提示 - history = json_post_list.get('history', []) # 获取请求中的历史记录 - - print(prompt) - messages = [ - {"role": "user", "content": prompt} - ] - - # 调用模型进行对话生成 - input_ids = tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') - output_ids = model.generate(input_ids.to('cuda'),max_new_tokens=2048) - - response = tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) - - - now = datetime.datetime.now() # 获取当前时间 - time = now.strftime("%Y-%m-%d %H:%M:%S") # 格式化时间为字符串 - # 构建响应JSON - answer = { - "response": response, - "status": 200, - "time": time - } - # 构建日志信息 - log = "[" + time + "] " + '", prompt:"' + prompt + '", response:"' + repr(response) + '"' - print(log) # 打印日志 - torch_gc() # 执行GPU内存清理 - return answer # 返回响应 - -# 主函数入口 -if __name__ == '__main__': - # 加载预训练的分词器和模型 - model_name_or_path = '/root/autodl-tmp/phi3/model/LLM-Research/Phi-3-mini-4k-instruct' - tokenizer = AutoTokenizer.from_pretrained(model_name_or_path) - model = AutoModelForCausalLM.from_pretrained(model_name_or_path, - device_map="cuda", - torch_dtype="auto", - trust_remote_code=True, - ).eval() - - # 启动FastAPI应用 - # 用6006端口可以将autodl的端口映射到本地,从而在本地使用api - uvicorn.run(app, host='0.0.0.0', port=6006, workers=1) # 在指定端口和主机上启动应用 -``` - -默认部署在 6006 端口,通过 POST 方法进行调用,可以使用curl调用,如下所示: - -```bash -curl -X POST "http://127.0.0.1:6006" \ - -H 'Content-Type: application/json' \ - -d '{"prompt": "你好", "history": []}' -``` -响应如下: -```json -{ - "response": "你好!如果你需要帮助或者有任何问题,请随时告诉我。", - "status": 200, - "time": "2024-05-09 16:36:43" -} -``` - -SSH端口映射 - -```bash -ssh -CNg -L 6006:127.0.0.1:6006 -p 【你的autodl机器的ssh端口】 root@[你的autodl机器地址] -ssh -CNg -L 6006:127.0.0.1:6006 -p 36494 root@region-45.autodl.pro -``` - -端口映射后,用postman访问 - +# Phi-3-mini-4k-instruct FastApi 部署调用 + +## 环境准备 + +在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 。 + +接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 + +![机器配置选择](../InternLM2/images/1.png) + +### 创建工作目录 + +创建本次phi3实践的工作目录`/root/autodl-tmp/phi3` + +```bash +# 创建工作目录 +mkdir -p /root/autodl-tmp/phi3 +``` + +### 安装依赖 + +```bash +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install fastapi==0.104.1 +pip install uvicorn==0.24.0.post1 +pip install requests==2.25.1 +pip install modelscope==1.9.5 +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.24.1 +``` + +由于phi3要求的transformers的版本为`4.41.0.dev0版本`。 + +各位可以先通过下面命令查看你的Transformers包的版本 + +```bash +pip list |grep transformers +``` + +如果版本不对,可以通过下面命令升级 + +```bash +# phi3升级transformers为4.41.0.dev0版本 +pip uninstall -y transformers && pip install git+https://github.com/huggingface/transformers +``` + + + +## 模型下载 + +使用 modelscope 中的`napshot_download`函数下载模型,第一个参数为模型名称,参数`cache_dir`为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建`download.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行`python /root/autodl-tmp/download.py`执行下载,模型大小为 8 GB,下载模型大概需要 10~15 分钟 + +```python +#模型下载 +from modelscope import snapshot_download +model_dir = snapshot_download('LLM-Research/Phi-3-mini-4k-instruct', cache_dir='/root/autodl-tmp/phi3', revision='master') +``` + +## 代码准备 + +在`/root/autodl-tmp`路径下新建`api.py`文件并在其中输入以下内容,粘贴代码后记得保存文件。下面的代码有很详细的注释,大家如有不理解的地方,欢迎提出issue。 + +```python +from fastapi import FastAPI, Request +from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig +import uvicorn +import json +import datetime +import torch + +# 设置设备参数 +DEVICE = "cuda" # 使用CUDA +DEVICE_ID = "0" # CUDA设备ID,如果未设置则为空 +CUDA_DEVICE = f"{DEVICE}:{DEVICE_ID}" if DEVICE_ID else DEVICE # 组合CUDA设备信息 + +# 清理GPU内存函数 +def torch_gc(): + if torch.cuda.is_available(): # 检查是否可用CUDA + with torch.cuda.device(CUDA_DEVICE): # 指定CUDA设备 + torch.cuda.empty_cache() # 清空CUDA缓存 + torch.cuda.ipc_collect() # 收集CUDA内存碎片 + +# 创建FastAPI应用 +app = FastAPI() + +# 处理POST请求的端点 +@app.post("/") +async def create_item(request: Request): + global model, tokenizer # 声明全局变量以便在函数内部使用模型和分词器 + json_post_raw = await request.json() # 获取POST请求的JSON数据 + json_post = json.dumps(json_post_raw) # 将JSON数据转换为字符串 + json_post_list = json.loads(json_post) # 将字符串转换为Python对象 + prompt = json_post_list.get('prompt') # 获取请求中的提示 + history = json_post_list.get('history', []) # 获取请求中的历史记录 + + print(prompt) + messages = [ + {"role": "user", "content": prompt} + ] + + # 调用模型进行对话生成 + input_ids = tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') + output_ids = model.generate(input_ids.to('cuda'),max_new_tokens=2048) + + response = tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) + + + now = datetime.datetime.now() # 获取当前时间 + time = now.strftime("%Y-%m-%d %H:%M:%S") # 格式化时间为字符串 + # 构建响应JSON + answer = { + "response": response, + "status": 200, + "time": time + } + # 构建日志信息 + log = "[" + time + "] " + '", prompt:"' + prompt + '", response:"' + repr(response) + '"' + print(log) # 打印日志 + torch_gc() # 执行GPU内存清理 + return answer # 返回响应 + +# 主函数入口 +if __name__ == '__main__': + # 加载预训练的分词器和模型 + model_name_or_path = '/root/autodl-tmp/phi3/model/LLM-Research/Phi-3-mini-4k-instruct' + tokenizer = AutoTokenizer.from_pretrained(model_name_or_path) + model = AutoModelForCausalLM.from_pretrained(model_name_or_path, + device_map="cuda", + torch_dtype="auto", + trust_remote_code=True, + ).eval() + + # 启动FastAPI应用 + # 用6006端口可以将autodl的端口映射到本地,从而在本地使用api + uvicorn.run(app, host='0.0.0.0', port=6006, workers=1) # 在指定端口和主机上启动应用 +``` + +默认部署在 6006 端口,通过 POST 方法进行调用,可以使用curl调用,如下所示: + +```bash +curl -X POST "http://127.0.0.1:6006" \ + -H 'Content-Type: application/json' \ + -d '{"prompt": "你好", "history": []}' +``` +响应如下: +```json +{ + "response": "你好!如果你需要帮助或者有任何问题,请随时告诉我。", + "status": 200, + "time": "2024-05-09 16:36:43" +} +``` + +SSH端口映射 + +```bash +ssh -CNg -L 6006:127.0.0.1:6006 -p 【你的autodl机器的ssh端口】 root@[你的autodl机器地址] +ssh -CNg -L 6006:127.0.0.1:6006 -p 36494 root@region-45.autodl.pro +``` + +端口映射后,用postman访问 + ![phi3-fastapi](./assets/01-1.png) \ No newline at end of file diff --git a/phi-3/02-Phi-3-mini-4k-instruct langchain 接入.md b/models/phi-3/02-Phi-3-mini-4k-instruct langchain 接入.md similarity index 97% rename from phi-3/02-Phi-3-mini-4k-instruct langchain 接入.md rename to models/phi-3/02-Phi-3-mini-4k-instruct langchain 接入.md index e23e76e..eb67e7a 100644 --- a/phi-3/02-Phi-3-mini-4k-instruct langchain 接入.md +++ b/models/phi-3/02-Phi-3-mini-4k-instruct langchain 接入.md @@ -1,141 +1,141 @@ -# Phi-3-mini-4k-instruct langchain 接入 - -## 环境准备 - -在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 。 - -接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 - -![机器配置选择](../InternLM2/images/1.png) - -### 创建工作目录 - -创建本次phi3实践的工作目录`/root/autodl-tmp/phi3` - -```bash -# 创建工作目录 -mkdir -p /root/autodl-tmp/phi3 -``` - -### 安装依赖 - -```bash -# 升级pip -python -m pip install --upgrade pip -# 更换 pypi 源加速库的安装 -pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple - -pip install fastapi==0.104.1 -pip install uvicorn==0.24.0.post1 -pip install requests==2.25.1 -pip install modelscope==1.9.5 -pip install streamlit==1.24.0 -pip install sentencepiece==0.1.99 -pip install accelerate==0.24.1 -pip install langchain==0.1.15 -``` - -由于phi3要求的transformers的版本为`4.41.0.dev0版本`。 - -各位可以先通过下面命令查看你的Transformers包的版本 - -```bash -pip list |grep transformers -``` - -如果版本不对,可以通过下面命令升级 - -```bash -# phi3升级transformers为4.41.0.dev0版本 -pip uninstall -y transformers && pip install git+https://github.com/huggingface/transformers -``` - - - -## 模型下载 - -使用 modelscope 中的`napshot_download`函数下载模型,第一个参数为模型名称,参数`cache_dir`为模型的下载路径。 - -在 /root/autodl-tmp 路径下新建`download.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行`python /root/autodl-tmp/download.py`执行下载,模型大小为 8 GB,下载模型大概需要 10~15 分钟 - -```python -#模型下载 -from modelscope import snapshot_download -model_dir = snapshot_download('LLM-Research/Phi-3-mini-4k-instruct', cache_dir='/root/autodl-tmp/phi3', revision='master') -``` - -## 代码准备 - -为便捷构建 LLM 应用,我们需要基于本地部署的 Phi-3-mini-4k-instruct,自定义一个 LLM 类,将 Phi-3-mini-4k-instruct 接入到 LangChain 框架中。完成自定义 LLM 类之后,可以以完全一致的方式调用 LangChain 的接口,而无需考虑底层模型调用的不一致。 - -基于本地部署的 Phi-3-mini-4k-instruct 自定义 LLM 类并不复杂,我们只需从 LangChain.llms.base.LLM 类继承一个子类,并重写构造函数与 _call 函数即可。 - -我们新建一个py文件`Phi3MiniLLM.py`,写入以下内容: - -```python -from langchain.llms.base import LLM -from typing import Any, List, Optional -from langchain.callbacks.manager import CallbackManagerForLLMRun -from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig -import torch - -class Phi3Mini_LLM(LLM): - # 基于本地 Phi-3-mini 自定义 LLM 类 - tokenizer: AutoTokenizer = None - model: AutoModelForCausalLM = None - - def __init__(self, mode_name_or_path :str): - - super().__init__() - print("正在从本地加载模型...") - self.tokenizer = AutoTokenizer.from_pretrained(mode_name_or_path, use_fast=False) - self.model = AutoModelForCausalLM.from_pretrained(mode_name_or_path, device_map="cuda", - torch_dtype="auto", - trust_remote_code=True,) - self.model.generation_config = GenerationConfig.from_pretrained(mode_name_or_path) - self.model.generation_config.pad_token_id = self.model.generation_config.eos_token_id - self.model = self.model.eval() - - print("完成本地模型的加载") - - - - def _call(self, prompt : str, stop: Optional[List[str]] = None, - run_manager: Optional[CallbackManagerForLLMRun] = None, - **kwargs: Any): - messages = [ - {"role": "user", "content": prompt} - ] - # 调用模型进行对话生成 - input_ids = self.tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') - output_ids = self.model.generate(input_ids.to('cuda'),max_new_tokens=2048) - - response = self.tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) - return response - - @property - def _llm_type(self) -> str: - return "Phi3Mini_LLM" -``` - -## 代码运行 - -然后就可以像使用任何其他的langchain大模型功能一样使用了。在jupyter上运行 - -```python -from Phi3MiniLLM import Phi3Mini_LLM -llm = Phi3Mini_LLM(mode_name_or_path = '/root/autodl-tmp/phi3/model/LLM-Research/Phi-3-mini-4k-instruct') -print(llm("你是谁")) -``` - -到这里,其实就已经把Phi-3-mini-4k-instruct 模型接入langchain了 - -![02-1](assets/02-1.png) - -通过langchain调用phi3-mini-4k-instruct 模型讲个故事 - -![02-2](assets/02-2.png) - -## TODO - +# Phi-3-mini-4k-instruct langchain 接入 + +## 环境准备 + +在 autodl 平台中租赁一个 3090 等 24G 显存的显卡机器,如下图所示镜像选择 PyTorch-->2.0.0-->3.8(ubuntu20.04)-->11.8 。 + +接下来打开刚刚租用服务器的 JupyterLab,并且打开其中的终端开始环境配置、模型下载和运行演示。 + +![机器配置选择](../InternLM2/images/1.png) + +### 创建工作目录 + +创建本次phi3实践的工作目录`/root/autodl-tmp/phi3` + +```bash +# 创建工作目录 +mkdir -p /root/autodl-tmp/phi3 +``` + +### 安装依赖 + +```bash +# 升级pip +python -m pip install --upgrade pip +# 更换 pypi 源加速库的安装 +pip config set global.index-url https://pypi.tuna.tsinghua.edu.cn/simple + +pip install fastapi==0.104.1 +pip install uvicorn==0.24.0.post1 +pip install requests==2.25.1 +pip install modelscope==1.9.5 +pip install streamlit==1.24.0 +pip install sentencepiece==0.1.99 +pip install accelerate==0.24.1 +pip install langchain==0.1.15 +``` + +由于phi3要求的transformers的版本为`4.41.0.dev0版本`。 + +各位可以先通过下面命令查看你的Transformers包的版本 + +```bash +pip list |grep transformers +``` + +如果版本不对,可以通过下面命令升级 + +```bash +# phi3升级transformers为4.41.0.dev0版本 +pip uninstall -y transformers && pip install git+https://github.com/huggingface/transformers +``` + + + +## 模型下载 + +使用 modelscope 中的`napshot_download`函数下载模型,第一个参数为模型名称,参数`cache_dir`为模型的下载路径。 + +在 /root/autodl-tmp 路径下新建`download.py` 文件并在其中输入以下内容,粘贴代码后记得保存文件,如下图所示。并运行`python /root/autodl-tmp/download.py`执行下载,模型大小为 8 GB,下载模型大概需要 10~15 分钟 + +```python +#模型下载 +from modelscope import snapshot_download +model_dir = snapshot_download('LLM-Research/Phi-3-mini-4k-instruct', cache_dir='/root/autodl-tmp/phi3', revision='master') +``` + +## 代码准备 + +为便捷构建 LLM 应用,我们需要基于本地部署的 Phi-3-mini-4k-instruct,自定义一个 LLM 类,将 Phi-3-mini-4k-instruct 接入到 LangChain 框架中。完成自定义 LLM 类之后,可以以完全一致的方式调用 LangChain 的接口,而无需考虑底层模型调用的不一致。 + +基于本地部署的 Phi-3-mini-4k-instruct 自定义 LLM 类并不复杂,我们只需从 LangChain.llms.base.LLM 类继承一个子类,并重写构造函数与 _call 函数即可。 + +我们新建一个py文件`Phi3MiniLLM.py`,写入以下内容: + +```python +from langchain.llms.base import LLM +from typing import Any, List, Optional +from langchain.callbacks.manager import CallbackManagerForLLMRun +from transformers import AutoTokenizer, AutoModelForCausalLM, GenerationConfig +import torch + +class Phi3Mini_LLM(LLM): + # 基于本地 Phi-3-mini 自定义 LLM 类 + tokenizer: AutoTokenizer = None + model: AutoModelForCausalLM = None + + def __init__(self, mode_name_or_path :str): + + super().__init__() + print("正在从本地加载模型...") + self.tokenizer = AutoTokenizer.from_pretrained(mode_name_or_path, use_fast=False) + self.model = AutoModelForCausalLM.from_pretrained(mode_name_or_path, device_map="cuda", + torch_dtype="auto", + trust_remote_code=True,) + self.model.generation_config = GenerationConfig.from_pretrained(mode_name_or_path) + self.model.generation_config.pad_token_id = self.model.generation_config.eos_token_id + self.model = self.model.eval() + + print("完成本地模型的加载") + + + + def _call(self, prompt : str, stop: Optional[List[str]] = None, + run_manager: Optional[CallbackManagerForLLMRun] = None, + **kwargs: Any): + messages = [ + {"role": "user", "content": prompt} + ] + # 调用模型进行对话生成 + input_ids = self.tokenizer.apply_chat_template(conversation=messages, tokenize=True, add_generation_prompt=True, return_tensors='pt') + output_ids = self.model.generate(input_ids.to('cuda'),max_new_tokens=2048) + + response = self.tokenizer.decode(output_ids[0][input_ids.shape[1]:], skip_special_tokens=True) + return response + + @property + def _llm_type(self) -> str: + return "Phi3Mini_LLM" +``` + +## 代码运行 + +然后就可以像使用任何其他的langchain大模型功能一样使用了。在jupyter上运行 + +```python +from Phi3MiniLLM import Phi3Mini_LLM +llm = Phi3Mini_LLM(mode_name_or_path = '/root/autodl-tmp/phi3/model/LLM-Research/Phi-3-mini-4k-instruct') +print(llm("你是谁")) +``` + +到这里,其实就已经把Phi-3-mini-4k-instruct 模型接入langchain了 + +![02-1](assets/02-1.png) + +通过langchain调用phi3-mini-4k-instruct 模型讲个故事 + +![02-2](assets/02-2.png) + +## TODO + 构建本地知识库数据。通过langchain搭建本地知识库小助手。 \ No newline at end of file diff --git a/phi-3/03-Phi-3-mini-4k-instruct WebDemo部署.md b/models/phi-3/03-Phi-3-mini-4k-instruct WebDemo部署.md similarity index 100% rename from phi-3/03-Phi-3-mini-4k-instruct WebDemo部署.md rename to models/phi-3/03-Phi-3-mini-4k-instruct WebDemo部署.md diff --git a/phi-3/04-Phi-3-mini-4k-Instruct Lora 微调.md b/models/phi-3/04-Phi-3-mini-4k-Instruct Lora 微调.md similarity index 100% rename from phi-3/04-Phi-3-mini-4k-Instruct Lora 微调.md rename to models/phi-3/04-Phi-3-mini-4k-Instruct Lora 微调.md diff --git a/phi-3/Phi-3-mini-4k-Instruct-Lora.ipynb b/models/phi-3/Phi-3-mini-4k-Instruct-Lora.ipynb similarity index 100% rename from phi-3/Phi-3-mini-4k-Instruct-Lora.ipynb rename to models/phi-3/Phi-3-mini-4k-Instruct-Lora.ipynb diff --git a/phi-3/assets/01-1.png b/models/phi-3/assets/01-1.png similarity index 100% rename from phi-3/assets/01-1.png rename to models/phi-3/assets/01-1.png diff --git a/phi-3/assets/02-1.png b/models/phi-3/assets/02-1.png similarity index 100% rename from phi-3/assets/02-1.png rename to models/phi-3/assets/02-1.png diff --git a/phi-3/assets/02-2.png b/models/phi-3/assets/02-2.png similarity index 100% rename from phi-3/assets/02-2.png rename to models/phi-3/assets/02-2.png diff --git a/phi-3/assets/03-1.png b/models/phi-3/assets/03-1.png similarity index 100% rename from phi-3/assets/03-1.png rename to models/phi-3/assets/03-1.png