From b29badc176435cfb679c383774f9e4a2b7cf7302 Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Wed, 31 Jul 2024 16:15:33 +0800 Subject: [PATCH 1/9] # add table recognition using struct-eqtable ## Changelog 31/07/20204 - Support table recognition. Table images will be converted into html. ### how to use the new feature: set the attribute 'table-mode' to 'true' in magic-pdf.json ### caution: it takes 200s to 500s to convert a single table image using cpu --- magic-pdf.template.json | 3 +- magic_pdf/dict2md/ocr_mkcontent.py | 9 +++- magic_pdf/libs/config_reader.py | 17 +++++++ .../model/doc_analyze_by_custom_model.py | 10 +++- magic_pdf/model/magic_model.py | 8 ++++ magic_pdf/model/pdf_extract_kit.py | 47 +++++++++++++++++++ .../structeqtable/StructTableModel.py | 20 ++++++++ .../pek_sub_modules/structeqtable/__init__.py | 0 .../resources/model_config/model_configs.yaml | 2 + 9 files changed, 112 insertions(+), 4 deletions(-) create mode 100644 magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py create mode 100644 magic_pdf/model/pek_sub_modules/structeqtable/__init__.py diff --git a/magic-pdf.template.json b/magic-pdf.template.json index 2c0223db..dacc59f1 100644 --- a/magic-pdf.template.json +++ b/magic-pdf.template.json @@ -5,5 +5,6 @@ }, "temp-output-dir":"/tmp", "models-dir":"/tmp/models", - "device-mode":"cpu" + "device-mode":"cpu", + "table-mode":"false" } \ No newline at end of file diff --git a/magic_pdf/dict2md/ocr_mkcontent.py b/magic_pdf/dict2md/ocr_mkcontent.py index eaadcdeb..afc5b174 100644 --- a/magic_pdf/dict2md/ocr_mkcontent.py +++ b/magic_pdf/dict2md/ocr_mkcontent.py @@ -128,7 +128,11 @@ def ocr_mk_markdown_with_para_core_v2(paras_of_layout, mode, img_buket_path=""): for line in block['lines']: for span in line['spans']: if span['type'] == ContentType.Table: - para_text += f"\n![]({join_path(img_buket_path, span['image_path'])}) \n" + # if processed by table model + if span.get('content', ''): + para_text += f"\n {span['content']} \n" + else: + para_text += f"\n![]({join_path(img_buket_path, span['image_path'])}) \n" for block in para_block['blocks']: # 3rd.拼table_footnote if block['type'] == BlockType.TableFootnote: para_text += merge_para_with_text(block) @@ -244,6 +248,9 @@ def para_to_standard_format_v2(para_block, img_buket_path): } for block in para_block['blocks']: if block['type'] == BlockType.TableBody: + #TODO + if block["lines"][0]["spans"][0].get('content', ''): + para_content['table_body'] = f"\n {block['lines'][0]['spans'][0]['content']} \n" para_content['img_path'] = join_path(img_buket_path, block["lines"][0]["spans"][0]['image_path']) if block['type'] == BlockType.TableCaption: para_content['table_caption'] = merge_para_with_text(block) diff --git a/magic_pdf/libs/config_reader.py b/magic_pdf/libs/config_reader.py index 7454c577..21b80ac1 100644 --- a/magic_pdf/libs/config_reader.py +++ b/magic_pdf/libs/config_reader.py @@ -86,6 +86,23 @@ def get_device(): else: return device +def get_table_mode(): + config = read_config() + table_mode = config.get("table-mode") + if table_mode is None: + logger.warning(f"'table-mode' not found in {CONFIG_FILE_NAME}, use 'False' as default") + return False + else: + table_mode = table_mode.lower() + if table_mode == "true": + boolean_value = True + elif table_mode == "False": + boolean_value = False + else: + logger.warning(f"invalid 'table-mode' value in {CONFIG_FILE_NAME}, use 'False' as default") + boolean_value = False + return boolean_value + if __name__ == "__main__": ak, sk, endpoint = get_s3_config("llm-raw") diff --git a/magic_pdf/model/doc_analyze_by_custom_model.py b/magic_pdf/model/doc_analyze_by_custom_model.py index d6916e74..f63f94f4 100644 --- a/magic_pdf/model/doc_analyze_by_custom_model.py +++ b/magic_pdf/model/doc_analyze_by_custom_model.py @@ -4,7 +4,7 @@ import fitz import numpy as np from loguru import logger -from magic_pdf.libs.config_reader import get_local_models_dir, get_device +from magic_pdf.libs.config_reader import get_local_models_dir, get_device, get_table_mode from magic_pdf.model.model_list import MODEL import magic_pdf.model as model_config @@ -82,7 +82,13 @@ def custom_model_init(ocr: bool = False, show_log: bool = False): # 从配置文件读取model-dir和device local_models_dir = get_local_models_dir() device = get_device() - custom_model = CustomPEKModel(ocr=ocr, show_log=show_log, models_dir=local_models_dir, device=device) + table_mode = get_table_mode() + model_input = {"ocr": ocr, + "show_log": show_log, + "models_dir": local_models_dir, + "device": device, + "table_mode": table_mode} + custom_model = CustomPEKModel(**model_input) else: logger.error("Not allow model_name!") exit(1) diff --git a/magic_pdf/model/magic_model.py b/magic_pdf/model/magic_model.py index 3e909fea..fec19294 100644 --- a/magic_pdf/model/magic_model.py +++ b/magic_pdf/model/magic_model.py @@ -560,6 +560,14 @@ class MagicModel: if category_id == 3: span["type"] = ContentType.Image elif category_id == 5: + # 获取table模型结果 + html = layout_det.get("html", None) + latex = layout_det.get("latex", None) + if html: + span["content"] = html + elif latex: + span["content"] = latex + span["type"] = ContentType.Table elif category_id == 13: span["content"] = layout_det["latex"] diff --git a/magic_pdf/model/pdf_extract_kit.py b/magic_pdf/model/pdf_extract_kit.py index 5b7b15e9..b559e253 100644 --- a/magic_pdf/model/pdf_extract_kit.py +++ b/magic_pdf/model/pdf_extract_kit.py @@ -1,6 +1,7 @@ from loguru import logger import os import time +from pypandoc import convert_text os.environ['NO_ALBUMENTATIONS_UPDATE'] = '1' # 禁止albumentations检查更新 try: @@ -10,6 +11,7 @@ try: import numpy as np import torch import torchtext + if torchtext.__version__ >= "0.18.0": torchtext.disable_torchtext_deprecation_warning() from PIL import Image @@ -30,6 +32,12 @@ except ImportError as e: from magic_pdf.model.pek_sub_modules.layoutlmv3.model_init import Layoutlmv3_Predictor from magic_pdf.model.pek_sub_modules.post_process import get_croped_image, latex_rm_whitespace from magic_pdf.model.pek_sub_modules.self_modify import ModifiedPaddleOCR +from magic_pdf.model.pek_sub_modules.structeqtable.StructTableModel import StructTableModel + + +def table_model_init(model_path): + table_model = StructTableModel(model_path) + return table_model def mfd_model_init(weight): @@ -95,6 +103,7 @@ class CustomPEKModel: # 初始化解析配置 self.apply_layout = kwargs.get("apply_layout", self.configs["config"]["layout"]) self.apply_formula = kwargs.get("apply_formula", self.configs["config"]["formula"]) + self.apply_table = kwargs.get("table_mode", self.configs["config"]["table"]) self.apply_ocr = ocr logger.info( "DocAnalysis init, this may take some times. apply_layout: {}, apply_formula: {}, apply_ocr: {}".format( @@ -129,6 +138,9 @@ class CustomPEKModel: if self.apply_ocr: self.ocr_model = ModifiedPaddleOCR(show_log=show_log) + # init structeqtable + if self.apply_table: + self.table_model = table_model_init(str(os.path.join(models_dir, self.configs["weights"]["table"]))) logger.info('DocAnalysis init done!') def __call__(self, image): @@ -249,4 +261,39 @@ class CustomPEKModel: ocr_cost = round(time.time() - ocr_start, 2) logger.info(f"ocr cost: {ocr_cost}") + # 表格识别 table recognition + if self.apply_table: + pil_img = Image.fromarray(image) + for layout in layout_res: + if layout.get("category_id", -1) == 5: + poly = layout["poly"] + xmin, ymin = int(poly[0]), int(poly[1]) + xmax, ymax = int(poly[4]), int(poly[5]) + + paste_x = 50 + paste_y = 50 + # 创建一个宽高各多50的白色背景 create a whiteboard with 50 larger width and length + new_width = xmax - xmin + paste_x * 2 + new_height = ymax - ymin + paste_y * 2 + new_image = Image.new('RGB', (new_width, new_height), 'white') + + # 裁剪图像 crop image + crop_box = (xmin, ymin, xmax, ymax) + cropped_img = pil_img.crop(crop_box) + new_image.paste(cropped_img, (paste_x, paste_y)) + start_time = time.time() + print("------------------table recognition processing begins-----------------") + latex_code = self.table_model.image2latex(new_image)[0] + end_time = time.time() + run_time = end_time - start_time + print(f"------------table recognition processing ends within {run_time}s-----") + + # try to convert latex to html + try: + html_code = convert_text(latex_code, 'html', format='latex') + layout["html"] = html_code + except Exception as e: + layout["latex"] = latex_code + logger.error(f"[pdf_extract_kit][CustomPEKModel]: converting latex to html failed: {e}") + return layout_res diff --git a/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py b/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py new file mode 100644 index 00000000..0fd1b902 --- /dev/null +++ b/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py @@ -0,0 +1,20 @@ +from struct_eqtable.model import StructTable +from pypandoc import convert_text +class StructTableModel: + def __init__(self, model_path, max_new_tokens=2048, max_time=400): + # init + self.model_path = model_path + self.max_new_tokens = max_new_tokens # maximum output tokens length + self.max_time = max_time # timeout for processing in seconds + self.model = StructTable(self.model_path, self.max_new_tokens, self.max_time) + + + def image2latex(self, image) -> str: + # + table_latex = self.model.forward(image) + return table_latex + + def image2html(self, image) -> str: + table_latex = self.image2latex(image) + table_html = convert_text(table_latex, 'html', format='latex') + return table_html diff --git a/magic_pdf/model/pek_sub_modules/structeqtable/__init__.py b/magic_pdf/model/pek_sub_modules/structeqtable/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/magic_pdf/resources/model_config/model_configs.yaml b/magic_pdf/resources/model_config/model_configs.yaml index 44cc8889..fb940f84 100644 --- a/magic_pdf/resources/model_config/model_configs.yaml +++ b/magic_pdf/resources/model_config/model_configs.yaml @@ -2,8 +2,10 @@ config: device: cpu layout: True formula: True + table: False weights: layout: Layout/model_final.pth mfd: MFD/weights.pt mfr: MFR/UniMERNet + table: Table/ \ No newline at end of file From d6c58ecca2683a38031323a458cd25858335cc7d Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Wed, 31 Jul 2024 20:02:29 +0800 Subject: [PATCH 2/9] # add table recognition using struct-eqtable ## Changelog 31/07/20204 - Support table recognition. Table images will be converted into LaTex. ### how to use the new feature: set the attribute 'table-mode' to 'true' in magic-pdf.json ### caution: it takes 200s to 500s to convert a single table image using cpu --- magic_pdf/dict2md/ocr_mkcontent.py | 2 +- magic_pdf/model/magic_model.py | 6 +----- magic_pdf/model/pdf_extract_kit.py | 8 +------- 3 files changed, 3 insertions(+), 13 deletions(-) diff --git a/magic_pdf/dict2md/ocr_mkcontent.py b/magic_pdf/dict2md/ocr_mkcontent.py index 5465c799..880d7f60 100644 --- a/magic_pdf/dict2md/ocr_mkcontent.py +++ b/magic_pdf/dict2md/ocr_mkcontent.py @@ -130,7 +130,7 @@ def ocr_mk_markdown_with_para_core_v2(paras_of_layout, mode, img_buket_path=""): if span['type'] == ContentType.Table: # if processed by table model if span.get('content', ''): - para_text += f"\n {span['content']} \n" + para_text += f"\n\n$\n {span['content']}\n$\n\n" else: para_text += f"\n![]({join_path(img_buket_path, span['image_path'])}) \n" for block in para_block['blocks']: # 3rd.拼table_footnote diff --git a/magic_pdf/model/magic_model.py b/magic_pdf/model/magic_model.py index fec19294..970bd079 100644 --- a/magic_pdf/model/magic_model.py +++ b/magic_pdf/model/magic_model.py @@ -561,13 +561,9 @@ class MagicModel: span["type"] = ContentType.Image elif category_id == 5: # 获取table模型结果 - html = layout_det.get("html", None) latex = layout_det.get("latex", None) - if html: - span["content"] = html - elif latex: + if latex: span["content"] = latex - span["type"] = ContentType.Table elif category_id == 13: span["content"] = layout_det["latex"] diff --git a/magic_pdf/model/pdf_extract_kit.py b/magic_pdf/model/pdf_extract_kit.py index b559e253..06e58e32 100644 --- a/magic_pdf/model/pdf_extract_kit.py +++ b/magic_pdf/model/pdf_extract_kit.py @@ -287,13 +287,7 @@ class CustomPEKModel: end_time = time.time() run_time = end_time - start_time print(f"------------table recognition processing ends within {run_time}s-----") + layout["latex"] = latex_code - # try to convert latex to html - try: - html_code = convert_text(latex_code, 'html', format='latex') - layout["html"] = html_code - except Exception as e: - layout["latex"] = latex_code - logger.error(f"[pdf_extract_kit][CustomPEKModel]: converting latex to html failed: {e}") return layout_res From d04f3f22f59a340f6f0546dea278a16376adcab8 Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Thu, 1 Aug 2024 14:41:36 +0800 Subject: [PATCH 3/9] # feat(model inference): add table recognition and convertion to LaTeX MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit # What's Changed ### New Features - Add table content recognition, we use weights of [StructEqTable](https://github.com/UniModal4Reasoning/StructEqTable-Deploy) to convert table image to LaTex. ### Instruction - pip install pypandoc struct-eqtable==0.1.0 - Download [StructEqTable weights](https://huggingface.co/wanderkid/PDF-Extract-Kit/tree/main/models/TabRec) and put it under models/ directory. - Edit 'table-mode' value to turn on table recognition function which is turned off by default. - If you did not download any models before, refer to [how to download models](docs/how_to_download_models_zh_cn.md)。 --- README_zh-CN_v2.md | 4 +++- docs/how_to_download_models_zh_cn.md | 10 ++++++++++ magic_pdf/dict2md/ocr_mkcontent.py | 3 +-- magic_pdf/resources/model_config/model_configs.yaml | 2 +- requirements-qa.txt | 3 ++- requirements.txt | 2 ++ 6 files changed, 19 insertions(+), 5 deletions(-) diff --git a/README_zh-CN_v2.md b/README_zh-CN_v2.md index bfbfc41e..33511216 100644 --- a/README_zh-CN_v2.md +++ b/README_zh-CN_v2.md @@ -91,6 +91,7 @@ MinerU诞生于[书生-浦语](https://github.com/InternLM/InternLM)的预训练 - 保留原文档的结构,包括标题、段落、列表等 - 提取图像、图片标题、表格、表格标题 - 自动识别文档中的公式并将公式转换成latex +- 自动识别文档中的表格并将表格转换成latex - 乱码PDF自动检测并启用OCR - 支持CPU和GPU环境 - 支持windows/linux/mac平台 @@ -235,7 +236,7 @@ TODO - [ ] 正文中列表识别 - [ ] 正文中代码块识别 - [ ] 目录识别 -- [ ] 表格识别 +- [x] 表格识别 - [ ] 化学式识别 - [ ] 几何图形识别 @@ -270,6 +271,7 @@ The project currently leverages PyMuPDF to deliver advanced functionalities; how - [PyMuPDF](https://github.com/pymupdf/PyMuPDF) - [fast-langdetect](https://github.com/LlmKira/fast-langdetect) - [pdfminer.six](https://github.com/pdfminer/pdfminer.six) +- [StructEqTable](https://github.com/UniModal4Reasoning/StructEqTable-Deploy) # Citation diff --git a/docs/how_to_download_models_zh_cn.md b/docs/how_to_download_models_zh_cn.md index fd07619c..994d10f5 100644 --- a/docs/how_to_download_models_zh_cn.md +++ b/docs/how_to_download_models_zh_cn.md @@ -73,5 +73,15 @@ git clone https://www.modelscope.cn/wanderkid/PDF-Extract-Kit.git │ ├── README.md │ ├── tokenizer_config.json │ └── tokenizer.json +│── TabRec +│ └─StructEqTable +│ ├── config.json +│ ├── generation_config.json +│ ├── model.safetensors +│ ├── preprocessor_config.json +│ ├── special_tokens_map.json +│ ├── spiece.model +│ ├── tokenizer.json +│ └── tokenizer_config.json └── README.md ``` diff --git a/magic_pdf/dict2md/ocr_mkcontent.py b/magic_pdf/dict2md/ocr_mkcontent.py index 880d7f60..c414d91d 100644 --- a/magic_pdf/dict2md/ocr_mkcontent.py +++ b/magic_pdf/dict2md/ocr_mkcontent.py @@ -253,9 +253,8 @@ def para_to_standard_format_v2(para_block, img_buket_path, page_idx): } for block in para_block['blocks']: if block['type'] == BlockType.TableBody: - #TODO if block["lines"][0]["spans"][0].get('content', ''): - para_content['table_body'] = f"\n {block['lines'][0]['spans'][0]['content']} \n" + para_content['table_body'] = f"\n\n$\n {block['lines'][0]['spans'][0]['content']}\n$\n\n" para_content['img_path'] = join_path(img_buket_path, block["lines"][0]["spans"][0]['image_path']) if block['type'] == BlockType.TableCaption: para_content['table_caption'] = merge_para_with_text(block) diff --git a/magic_pdf/resources/model_config/model_configs.yaml b/magic_pdf/resources/model_config/model_configs.yaml index fb940f84..59143ddb 100644 --- a/magic_pdf/resources/model_config/model_configs.yaml +++ b/magic_pdf/resources/model_config/model_configs.yaml @@ -8,4 +8,4 @@ weights: layout: Layout/model_final.pth mfd: MFD/weights.pt mfr: MFR/UniMERNet - table: Table/ \ No newline at end of file + table: TabRec/StructEqTable \ No newline at end of file diff --git a/requirements-qa.txt b/requirements-qa.txt index d397f6cd..5c9dd1e3 100644 --- a/requirements-qa.txt +++ b/requirements-qa.txt @@ -13,4 +13,5 @@ scikit-learn tqdm htmltabletomd pypandoc -pyopenssl==24.0.0 \ No newline at end of file +pyopenssl==24.0.0 +struct-eqtable==0.1.0 \ No newline at end of file diff --git a/requirements.txt b/requirements.txt index ff17e4b6..a8efe0c1 100644 --- a/requirements.txt +++ b/requirements.txt @@ -8,4 +8,6 @@ fast-langdetect==0.2.0 wordninja>=2.0.0 scikit-learn>=1.0.2 pdfminer.six==20231228 +pypandoc +struct-eqtable==0.1.0 # The requirements.txt must ensure that only necessary external dependencies are introduced. If there are new dependencies to add, please contact the project administrator. From 4c096443c7e01b0eac3b28e2a94d6144f47e6f2c Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Thu, 1 Aug 2024 15:20:28 +0800 Subject: [PATCH 4/9] add table recognition and convertion to LaTeX --- magic_pdf/model/pdf_extract_kit.py | 6 +++--- .../pek_sub_modules/structeqtable/StructTableModel.py | 8 +++++--- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/magic_pdf/model/pdf_extract_kit.py b/magic_pdf/model/pdf_extract_kit.py index 06e58e32..56a08995 100644 --- a/magic_pdf/model/pdf_extract_kit.py +++ b/magic_pdf/model/pdf_extract_kit.py @@ -35,8 +35,8 @@ from magic_pdf.model.pek_sub_modules.self_modify import ModifiedPaddleOCR from magic_pdf.model.pek_sub_modules.structeqtable.StructTableModel import StructTableModel -def table_model_init(model_path): - table_model = StructTableModel(model_path) +def table_model_init(model_path, _device_ = 'cpu'): + table_model = StructTableModel(model_path, device = _device_) return table_model @@ -140,7 +140,7 @@ class CustomPEKModel: # init structeqtable if self.apply_table: - self.table_model = table_model_init(str(os.path.join(models_dir, self.configs["weights"]["table"]))) + self.table_model = table_model_init(str(os.path.join(models_dir, self.configs["weights"]["table"])), _device_=self.device) logger.info('DocAnalysis init done!') def __call__(self, image): diff --git a/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py b/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py index 0fd1b902..1f178f27 100644 --- a/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py +++ b/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py @@ -1,13 +1,15 @@ from struct_eqtable.model import StructTable from pypandoc import convert_text class StructTableModel: - def __init__(self, model_path, max_new_tokens=2048, max_time=400): + def __init__(self, model_path, max_new_tokens=2048, max_time=400, device = 'cpu'): # init self.model_path = model_path self.max_new_tokens = max_new_tokens # maximum output tokens length self.max_time = max_time # timeout for processing in seconds - self.model = StructTable(self.model_path, self.max_new_tokens, self.max_time) - + if device == 'cpu': + self.model = StructTable(self.model_path, self.max_new_tokens, self.max_time) + else: + self.model = StructTable(self.model_path, self.max_new_tokens, self.max_time).cuda() def image2latex(self, image) -> str: # From 78238f3989d558917e8517e0cab9a04645535865 Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Thu, 1 Aug 2024 16:41:48 +0800 Subject: [PATCH 5/9] add table recognition and conversion to LaTeX --- .../model/pek_sub_modules/structeqtable/StructTableModel.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py b/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py index 1f178f27..cfd9fa2d 100644 --- a/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py +++ b/magic_pdf/model/pek_sub_modules/structeqtable/StructTableModel.py @@ -6,10 +6,10 @@ class StructTableModel: self.model_path = model_path self.max_new_tokens = max_new_tokens # maximum output tokens length self.max_time = max_time # timeout for processing in seconds - if device == 'cpu': - self.model = StructTable(self.model_path, self.max_new_tokens, self.max_time) - else: + if device == 'cuda': self.model = StructTable(self.model_path, self.max_new_tokens, self.max_time).cuda() + else: + self.model = StructTable(self.model_path, self.max_new_tokens, self.max_time) def image2latex(self, image) -> str: # From dbe628ee0708ad662533ec2d97d03d30eee32e1e Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Thu, 1 Aug 2024 17:43:44 +0800 Subject: [PATCH 6/9] add table recognition and conversion to LaTeX --- magic-pdf.template.json | 5 ++++- magic_pdf/dict2md/ocr_mkcontent.py | 12 +++++++----- magic_pdf/libs/config_reader.py | 18 +++--------------- magic_pdf/model/doc_analyze_by_custom_model.py | 6 +++--- magic_pdf/model/magic_model.py | 2 +- magic_pdf/model/pdf_extract_kit.py | 17 +++++++++-------- .../resources/model_config/model_configs.yaml | 4 +++- 7 files changed, 30 insertions(+), 34 deletions(-) diff --git a/magic-pdf.template.json b/magic-pdf.template.json index dacc59f1..dacb6d92 100644 --- a/magic-pdf.template.json +++ b/magic-pdf.template.json @@ -6,5 +6,8 @@ "temp-output-dir":"/tmp", "models-dir":"/tmp/models", "device-mode":"cpu", - "table-mode":"false" + "table-config": { + "is_table_recog_enable": false, + "max_time": 400 + } } \ No newline at end of file diff --git a/magic_pdf/dict2md/ocr_mkcontent.py b/magic_pdf/dict2md/ocr_mkcontent.py index c414d91d..c3c08fa6 100644 --- a/magic_pdf/dict2md/ocr_mkcontent.py +++ b/magic_pdf/dict2md/ocr_mkcontent.py @@ -120,19 +120,21 @@ def ocr_mk_markdown_with_para_core_v2(paras_of_layout, mode, img_buket_path=""): if mode == 'nlp': continue elif mode == 'mm': + table_caption = '' for block in para_block['blocks']: # 1st.拼table_caption if block['type'] == BlockType.TableCaption: - para_text += merge_para_with_text(block) + table_caption = merge_para_with_text(block) + para_text += table_caption for block in para_block['blocks']: # 2nd.拼table_body if block['type'] == BlockType.TableBody: for line in block['lines']: for span in line['spans']: if span['type'] == ContentType.Table: # if processed by table model - if span.get('content', ''): - para_text += f"\n\n$\n {span['content']}\n$\n\n" + if span.get('latex', ''): + para_text += f"\n\n$\n {span['latex']}\n$\n\n" else: - para_text += f"\n![]({join_path(img_buket_path, span['image_path'])}) \n" + para_text += f"\n![{table_caption}]({join_path(img_buket_path, span['image_path'])}) \n" for block in para_block['blocks']: # 3rd.拼table_footnote if block['type'] == BlockType.TableFootnote: para_text += merge_para_with_text(block) @@ -253,7 +255,7 @@ def para_to_standard_format_v2(para_block, img_buket_path, page_idx): } for block in para_block['blocks']: if block['type'] == BlockType.TableBody: - if block["lines"][0]["spans"][0].get('content', ''): + if block["lines"][0]["spans"][0].get('latex', ''): para_content['table_body'] = f"\n\n$\n {block['lines'][0]['spans'][0]['content']}\n$\n\n" para_content['img_path'] = join_path(img_buket_path, block["lines"][0]["spans"][0]['image_path']) if block['type'] == BlockType.TableCaption: diff --git a/magic_pdf/libs/config_reader.py b/magic_pdf/libs/config_reader.py index 21b80ac1..34817efc 100644 --- a/magic_pdf/libs/config_reader.py +++ b/magic_pdf/libs/config_reader.py @@ -86,22 +86,10 @@ def get_device(): else: return device -def get_table_mode(): +def get_table_recog_config(): config = read_config() - table_mode = config.get("table-mode") - if table_mode is None: - logger.warning(f"'table-mode' not found in {CONFIG_FILE_NAME}, use 'False' as default") - return False - else: - table_mode = table_mode.lower() - if table_mode == "true": - boolean_value = True - elif table_mode == "False": - boolean_value = False - else: - logger.warning(f"invalid 'table-mode' value in {CONFIG_FILE_NAME}, use 'False' as default") - boolean_value = False - return boolean_value + table_config = config.get("table-config") + return table_config if __name__ == "__main__": diff --git a/magic_pdf/model/doc_analyze_by_custom_model.py b/magic_pdf/model/doc_analyze_by_custom_model.py index 3d0ed6da..6adc6ac0 100644 --- a/magic_pdf/model/doc_analyze_by_custom_model.py +++ b/magic_pdf/model/doc_analyze_by_custom_model.py @@ -4,7 +4,7 @@ import fitz import numpy as np from loguru import logger -from magic_pdf.libs.config_reader import get_local_models_dir, get_device, get_table_mode +from magic_pdf.libs.config_reader import get_local_models_dir, get_device, get_table_recog_config from magic_pdf.model.model_list import MODEL import magic_pdf.model as model_config @@ -84,12 +84,12 @@ def custom_model_init(ocr: bool = False, show_log: bool = False): # 从配置文件读取model-dir和device local_models_dir = get_local_models_dir() device = get_device() - table_mode = get_table_mode() + table_config = get_table_recog_config() model_input = {"ocr": ocr, "show_log": show_log, "models_dir": local_models_dir, "device": device, - "table_mode": table_mode} + "table_config": table_config} custom_model = CustomPEKModel(**model_input) else: logger.error("Not allow model_name!") diff --git a/magic_pdf/model/magic_model.py b/magic_pdf/model/magic_model.py index 970bd079..d6164bae 100644 --- a/magic_pdf/model/magic_model.py +++ b/magic_pdf/model/magic_model.py @@ -563,7 +563,7 @@ class MagicModel: # 获取table模型结果 latex = layout_det.get("latex", None) if latex: - span["content"] = latex + span["latex"] = latex span["type"] = ContentType.Table elif category_id == 13: span["content"] = layout_det["latex"] diff --git a/magic_pdf/model/pdf_extract_kit.py b/magic_pdf/model/pdf_extract_kit.py index 56a08995..55b93731 100644 --- a/magic_pdf/model/pdf_extract_kit.py +++ b/magic_pdf/model/pdf_extract_kit.py @@ -35,8 +35,8 @@ from magic_pdf.model.pek_sub_modules.self_modify import ModifiedPaddleOCR from magic_pdf.model.pek_sub_modules.structeqtable.StructTableModel import StructTableModel -def table_model_init(model_path, _device_ = 'cpu'): - table_model = StructTableModel(model_path, device = _device_) +def table_model_init(model_path, max_time=400, _device_='cpu'): + table_model = StructTableModel(model_path, max_time=max_time, device=_device_) return table_model @@ -103,7 +103,7 @@ class CustomPEKModel: # 初始化解析配置 self.apply_layout = kwargs.get("apply_layout", self.configs["config"]["layout"]) self.apply_formula = kwargs.get("apply_formula", self.configs["config"]["formula"]) - self.apply_table = kwargs.get("table_mode", self.configs["config"]["table"]) + self.table_config = kwargs.get("table_config", self.configs["config"]["table_config"]) self.apply_ocr = ocr logger.info( "DocAnalysis init, this may take some times. apply_layout: {}, apply_formula: {}, apply_ocr: {}".format( @@ -139,8 +139,10 @@ class CustomPEKModel: self.ocr_model = ModifiedPaddleOCR(show_log=show_log) # init structeqtable - if self.apply_table: - self.table_model = table_model_init(str(os.path.join(models_dir, self.configs["weights"]["table"])), _device_=self.device) + if self.table_config.get("is_table_recog_enable", False): + max_time = self.table_config.get("max_time", 400) + self.table_model = table_model_init(str(os.path.join(models_dir, self.configs["weights"]["table"])), + max_time=max_time, _device_=self.device) logger.info('DocAnalysis init done!') def __call__(self, image): @@ -282,12 +284,11 @@ class CustomPEKModel: cropped_img = pil_img.crop(crop_box) new_image.paste(cropped_img, (paste_x, paste_y)) start_time = time.time() - print("------------------table recognition processing begins-----------------") + logger.info("------------------table recognition processing begins-----------------") latex_code = self.table_model.image2latex(new_image)[0] end_time = time.time() run_time = end_time - start_time - print(f"------------table recognition processing ends within {run_time}s-----") + logger.info(f"------------table recognition processing ends within {run_time}s-----") layout["latex"] = latex_code - return layout_res diff --git a/magic_pdf/resources/model_config/model_configs.yaml b/magic_pdf/resources/model_config/model_configs.yaml index 59143ddb..1b5c3aae 100644 --- a/magic_pdf/resources/model_config/model_configs.yaml +++ b/magic_pdf/resources/model_config/model_configs.yaml @@ -2,7 +2,9 @@ config: device: cpu layout: True formula: True - table: False + table_config: + is_table_recog_enable: False + max_time: 400 weights: layout: Layout/model_final.pth From b9667fd3e3914b57ec332355813a526b4838ee6e Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Thu, 1 Aug 2024 18:45:56 +0800 Subject: [PATCH 7/9] add table recognition and conversion to LaTeX --- magic_pdf/dict2md/ocr_mkcontent.py | 1 - magic_pdf/model/pdf_extract_kit.py | 3 ++- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/magic_pdf/dict2md/ocr_mkcontent.py b/magic_pdf/dict2md/ocr_mkcontent.py index c3c08fa6..8702d054 100644 --- a/magic_pdf/dict2md/ocr_mkcontent.py +++ b/magic_pdf/dict2md/ocr_mkcontent.py @@ -124,7 +124,6 @@ def ocr_mk_markdown_with_para_core_v2(paras_of_layout, mode, img_buket_path=""): for block in para_block['blocks']: # 1st.拼table_caption if block['type'] == BlockType.TableCaption: table_caption = merge_para_with_text(block) - para_text += table_caption for block in para_block['blocks']: # 2nd.拼table_body if block['type'] == BlockType.TableBody: for line in block['lines']: diff --git a/magic_pdf/model/pdf_extract_kit.py b/magic_pdf/model/pdf_extract_kit.py index 55b93731..f4e6f9cd 100644 --- a/magic_pdf/model/pdf_extract_kit.py +++ b/magic_pdf/model/pdf_extract_kit.py @@ -104,6 +104,7 @@ class CustomPEKModel: self.apply_layout = kwargs.get("apply_layout", self.configs["config"]["layout"]) self.apply_formula = kwargs.get("apply_formula", self.configs["config"]["formula"]) self.table_config = kwargs.get("table_config", self.configs["config"]["table_config"]) + self.apply_table = self.table_config.get("is_table_recog_enable", False) self.apply_ocr = ocr logger.info( "DocAnalysis init, this may take some times. apply_layout: {}, apply_formula: {}, apply_ocr: {}".format( @@ -139,7 +140,7 @@ class CustomPEKModel: self.ocr_model = ModifiedPaddleOCR(show_log=show_log) # init structeqtable - if self.table_config.get("is_table_recog_enable", False): + if self.apply_table: max_time = self.table_config.get("max_time", 400) self.table_model = table_model_init(str(os.path.join(models_dir, self.configs["weights"]["table"])), max_time=max_time, _device_=self.device) From ec7271faee9f175507ccdbcfcd2f081836fbdc97 Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Mon, 5 Aug 2024 16:54:35 +0800 Subject: [PATCH 8/9] fix table recognition bug#321 --- magic_pdf/dict2md/ocr_mkcontent.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/magic_pdf/dict2md/ocr_mkcontent.py b/magic_pdf/dict2md/ocr_mkcontent.py index 2195f7c5..0cc887ce 100644 --- a/magic_pdf/dict2md/ocr_mkcontent.py +++ b/magic_pdf/dict2md/ocr_mkcontent.py @@ -255,7 +255,7 @@ def para_to_standard_format_v2(para_block, img_buket_path, page_idx): for block in para_block['blocks']: if block['type'] == BlockType.TableBody: if block["lines"][0]["spans"][0].get('latex', ''): - para_content['table_body'] = f"\n\n$\n {block['lines'][0]['spans'][0]['content']}\n$\n\n" + para_content['table_body'] = f"\n\n$\n {block['lines'][0]['spans'][0]['latex']}\n$\n\n" para_content['img_path'] = join_path(img_buket_path, block["lines"][0]["spans"][0]['image_path']) if block['type'] == BlockType.TableCaption: para_content['table_caption'] = merge_para_with_text(block) From cae215bb5e47ddf21211dfcab34cccb1791dc00b Mon Sep 17 00:00:00 2001 From: liukaiwen Date: Mon, 5 Aug 2024 17:01:12 +0800 Subject: [PATCH 9/9] fix table recognition bug#321 --- magic_pdf/libs/config_reader.py | 5 ----- magic_pdf/model/pdf_extract_kit.py | 1 - requirements.txt | 2 -- 3 files changed, 8 deletions(-) diff --git a/magic_pdf/libs/config_reader.py b/magic_pdf/libs/config_reader.py index 8006a4ab..eb282903 100644 --- a/magic_pdf/libs/config_reader.py +++ b/magic_pdf/libs/config_reader.py @@ -76,11 +76,6 @@ def get_device(): else: return device -def get_table_recog_config(): - config = read_config() - table_config = config.get("table-config") - return table_config - def get_table_recog_config(): config = read_config() diff --git a/magic_pdf/model/pdf_extract_kit.py b/magic_pdf/model/pdf_extract_kit.py index aa5c06a1..56b90b38 100644 --- a/magic_pdf/model/pdf_extract_kit.py +++ b/magic_pdf/model/pdf_extract_kit.py @@ -1,7 +1,6 @@ from loguru import logger import os import time -from pypandoc import convert_text os.environ['NO_ALBUMENTATIONS_UPDATE'] = '1' # 禁止albumentations检查更新 diff --git a/requirements.txt b/requirements.txt index a8efe0c1..ff17e4b6 100644 --- a/requirements.txt +++ b/requirements.txt @@ -8,6 +8,4 @@ fast-langdetect==0.2.0 wordninja>=2.0.0 scikit-learn>=1.0.2 pdfminer.six==20231228 -pypandoc -struct-eqtable==0.1.0 # The requirements.txt must ensure that only necessary external dependencies are introduced. If there are new dependencies to add, please contact the project administrator.