From d67af17f493413e6fb05fdfd1bd7f26b867e2bf1 Mon Sep 17 00:00:00 2001 From: quyuan Date: Sat, 13 Jul 2024 18:35:44 +0800 Subject: [PATCH] add ci --- tests/test_cli/lib/calculate_score.py | 2 +- tests/test_cli/lib/pre_clean.py | 3 --- 2 files changed, 1 insertion(+), 4 deletions(-) diff --git a/tests/test_cli/lib/calculate_score.py b/tests/test_cli/lib/calculate_score.py index 719adba4..a7c140d6 100644 --- a/tests/test_cli/lib/calculate_score.py +++ b/tests/test_cli/lib/calculate_score.py @@ -4,12 +4,12 @@ calculate_score import os import re import json +from Levenshtein import distance from lib import scoring from nltk.translate.bleu_score import sentence_bleu, SmoothingFunction from nltk.tokenize import word_tokenize import nltk nltk.download('punkt') -from Levenshtein import distance class Scoring: """ diff --git a/tests/test_cli/lib/pre_clean.py b/tests/test_cli/lib/pre_clean.py index ac15ba50..4f423021 100644 --- a/tests/test_cli/lib/pre_clean.py +++ b/tests/test_cli/lib/pre_clean.py @@ -118,9 +118,6 @@ def clean_data(prod_type, download_dir): with open(input_file, 'r', encoding='utf-8') as fr: content = fr.read() new_content = clean_markdown_images(content) - new_content = convert_html_table_to_md(new_content) - new_content = convert_latext_to_md(new_content) - new_content = convert_htmltale_to_md(new_content) with open(output_file, 'w', encoding='utf-8') as fw: fw.write(new_content)