mirror of
https://github.com/ooyinet/WeClone.git
synced 2026-08-31 00:50:06 +08:00
在pyproject.toml中添加langchain依赖,并在strategies.py中引入PromptTemplate以支持数据清洗功能的实现。
This commit is contained in:
@@ -16,6 +16,7 @@ dependencies = [
|
||||
"torch>=2.6.0",
|
||||
"transformers==4.49.0",
|
||||
"tomli; python_version < '3.11'",
|
||||
"langchain",
|
||||
]
|
||||
|
||||
[tool.weclone]
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict
|
||||
# from ..models import ChatMessage # 如果需要操作特定模型,取消注释并调整
|
||||
from langchain_core.prompts import PromptTemplate
|
||||
from weclone.prompts.clean_data import CLEAN_PROMPT
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -28,16 +29,9 @@ class CleaningStrategy(ABC):
|
||||
class LLMCleaningStrategy(CleaningStrategy):
|
||||
"""使用大模型进行数据清洗的策略"""
|
||||
|
||||
# 这里可以添加LLM相关的配置,例如模型名称、API密钥等
|
||||
# model_name: str = "your_llm_model"
|
||||
|
||||
def clean(self, data: Any) -> Any:
|
||||
"""
|
||||
使用大模型清洗数据。
|
||||
具体的实现需要根据您选择的LLM API和清洗任务来定。
|
||||
"""
|
||||
# 此处为调用LLM进行清洗的逻辑占位符
|
||||
print(f"使用LLM清洗数据: {data}")
|
||||
# 假设LLM返回了清洗后的数据
|
||||
cleaned_data = f"LLM cleaned: {data}" # 示例返回值
|
||||
return cleaned_data
|
||||
prompt_template = PromptTemplate.from_template(CLEAN_PROMPT)
|
||||
|
||||
prompt_template.invoke({"topic": "cats"})
|
||||
# prompt_template.
|
||||
# return cleaned_data
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
CLEAN_PROMPT = """
|
||||
请根据以下文本生成一个简洁的摘要:
|
||||
{text_input}
|
||||
摘要应包含关键信息,长度不超过三句话。
|
||||
"""
|
||||
Reference in New Issue
Block a user