Merge pull request #298 from Zeyi-Lin/master
docs: improve finetune code
@@ -4,7 +4,7 @@
|
||||
|
||||
Lora 是一种高效微调方法,深入了解其原理可参见博客:[知乎|深入浅出 Lora](https://zhuanlan.zhihu.com/p/650197598)。
|
||||
|
||||
训练过程:<a href="https://swanlab.cn/@ZeyiLin/Qwen2-VL-finetune/runs/53vm3y7sp5h5fzlmlc5up/chart" target="_blank">Qwen2-VL-finetune
|
||||
训练过程:<a href="https://swanlab.cn/@ZeyiLin/Qwen2-VL-finetune/runs/pkgest5xhdn3ukpdy6kv5/chart" target="_blank">Qwen2-VL-finetune
|
||||
</a>
|
||||
|
||||
## 目录
|
||||
@@ -56,7 +56,7 @@ pip install sentencepiece==0.2.0
|
||||
pip install accelerate==1.1.1
|
||||
pip install datasets==2.18.0
|
||||
pip install peft==0.13.2
|
||||
pip install swanlab==0.3.25
|
||||
pip install swanlab==0.3.27
|
||||
pip install qwen-vl-utils==0.0.8
|
||||
```
|
||||
|
||||
@@ -299,33 +299,67 @@ def process_func(example):
|
||||
conversation = example["conversations"]
|
||||
input_content = conversation[0]["value"]
|
||||
output_content = conversation[1]["value"]
|
||||
|
||||
instruction = tokenizer(
|
||||
f"<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n<|im_start|>user\n{input_content}<|im_end|>\n<|im_start|>assistant\n",
|
||||
add_special_tokens=False,
|
||||
file_path = input_content.split("<|vision_start|>")[1].split("<|vision_end|>")[0] # 获取图像路径
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image",
|
||||
"image": f"{file_path}",
|
||||
"resized_height": 280,
|
||||
"resized_width": 280,
|
||||
},
|
||||
{"type": "text", "text": "COCO Yes:"},
|
||||
],
|
||||
}
|
||||
]
|
||||
text = processor.apply_chat_template(
|
||||
messages, tokenize=False, add_generation_prompt=True
|
||||
) # 获取文本
|
||||
image_inputs, video_inputs = process_vision_info(messages) # 获取数据数据(预处理过)
|
||||
inputs = processor(
|
||||
text=[text],
|
||||
images=image_inputs,
|
||||
videos=video_inputs,
|
||||
padding=True,
|
||||
return_tensors="pt",
|
||||
)
|
||||
inputs = {key: value.tolist() for key, value in inputs.items()} #tensor -> list,为了方便拼接
|
||||
instruction = inputs
|
||||
|
||||
response = tokenizer(f"{output_content}", add_special_tokens=False)
|
||||
|
||||
|
||||
input_ids = (
|
||||
instruction["input_ids"] + response["input_ids"] + [tokenizer.pad_token_id]
|
||||
instruction["input_ids"][0] + response["input_ids"] + [tokenizer.pad_token_id]
|
||||
)
|
||||
attention_mask = instruction["attention_mask"] + response["attention_mask"] + [1]
|
||||
|
||||
attention_mask = instruction["attention_mask"][0] + response["attention_mask"] + [1]
|
||||
labels = (
|
||||
[-100] * len(instruction["input_ids"])
|
||||
+ response["input_ids"]
|
||||
+ [tokenizer.pad_token_id]
|
||||
[-100] * len(instruction["input_ids"][0])
|
||||
+ response["input_ids"]
|
||||
+ [tokenizer.pad_token_id]
|
||||
)
|
||||
|
||||
if len(input_ids) > MAX_LENGTH: # 做一个截断
|
||||
input_ids = input_ids[:MAX_LENGTH]
|
||||
attention_mask = attention_mask[:MAX_LENGTH]
|
||||
labels = labels[:MAX_LENGTH]
|
||||
|
||||
return {"input_ids": input_ids, "attention_mask": attention_mask, "labels": labels}
|
||||
|
||||
input_ids = torch.tensor(input_ids)
|
||||
attention_mask = torch.tensor(attention_mask)
|
||||
labels = torch.tensor(labels)
|
||||
inputs['pixel_values'] = torch.tensor(inputs['pixel_values'])
|
||||
inputs['image_grid_thw'] = torch.tensor(inputs['image_grid_thw']).squeeze(0) #由(1,h,w)变换为(h,w)
|
||||
return {"input_ids": input_ids, "attention_mask": attention_mask, "labels": labels,
|
||||
"pixel_values": inputs['pixel_values'], "image_grid_thw": inputs['image_grid_thw']}
|
||||
|
||||
|
||||
def predict(messages, model):
|
||||
# 准备推理
|
||||
text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
|
||||
text = processor.apply_chat_template(
|
||||
messages, tokenize=False, add_generation_prompt=True
|
||||
)
|
||||
image_inputs, video_inputs = process_vision_info(messages)
|
||||
inputs = processor(
|
||||
text=[text],
|
||||
@@ -395,6 +429,7 @@ args = TrainingArguments(
|
||||
per_device_train_batch_size=4,
|
||||
gradient_accumulation_steps=4,
|
||||
logging_steps=10,
|
||||
logging_first_step=5,
|
||||
num_train_epochs=2,
|
||||
save_steps=100,
|
||||
learning_rate=1e-4,
|
||||
@@ -460,12 +495,12 @@ for item in test_dataset:
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image",
|
||||
"image": origin_image_path
|
||||
"type": "image",
|
||||
"image": origin_image_path
|
||||
},
|
||||
{
|
||||
"type": "text",
|
||||
"text": "COCO Yes:"
|
||||
"type": "text",
|
||||
"text": "COCO Yes:"
|
||||
}
|
||||
]}]
|
||||
|
||||
@@ -581,6 +616,7 @@ print(output_text)
|
||||
使用4张A100 40GB显卡,batch size为4,gradient accumulation steps为4,训练2个epoch的用时为1分钟57秒。
|
||||
|
||||

|
||||

|
||||
|
||||
### 注意
|
||||
|
||||
|
||||
|
Before Width: | Height: | Size: 349 KiB After Width: | Height: | Size: 612 KiB |
|
Before Width: | Height: | Size: 4.2 MiB After Width: | Height: | Size: 2.3 MiB |
|
Before Width: | Height: | Size: 205 KiB After Width: | Height: | Size: 220 KiB |
|
After Width: | Height: | Size: 217 KiB |