From 61747bafddc6074d41a07e7ddc45f2e44b584ab7 Mon Sep 17 00:00:00 2001 From: myhloli Date: Fri, 31 Oct 2025 14:57:17 +0800 Subject: [PATCH 01/28] fix: center-align column header for vlm accuracy in index.md table --- docs/en/quick_start/index.md | 2 +- docs/zh/quick_start/index.md | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/docs/en/quick_start/index.md b/docs/en/quick_start/index.md index 8b77e986..cc349ba7 100644 --- a/docs/en/quick_start/index.md +++ b/docs/en/quick_start/index.md @@ -31,7 +31,7 @@ A WebUI developed based on Gradio, with a simple interface and only core parsing Parsing Backend pipeline
(Accuracy1 82+) - vlm (Accuracy1 90+) + vlm (Accuracy1 90+) transformers diff --git a/docs/zh/quick_start/index.md b/docs/zh/quick_start/index.md index 82b432df..d00258b3 100644 --- a/docs/zh/quick_start/index.md +++ b/docs/zh/quick_start/index.md @@ -26,12 +26,12 @@ > > 在非主线环境中,由于硬件、软件配置的多样性,以及第三方依赖项的兼容性问题,我们无法100%保证项目的完全可用性。因此,对于希望在非推荐环境中使用本项目的用户,我们建议先仔细阅读文档以及FAQ,大多数问题已经在FAQ中有对应的解决方案,除此之外我们鼓励社区反馈问题,以便我们能够逐步扩大支持范围。 - +
- + From cce16daf1f5a63eb422b44ae4353b3d2919c6db4 Mon Sep 17 00:00:00 2001 From: myhloli Date: Fri, 31 Oct 2025 15:37:08 +0800 Subject: [PATCH 02/28] fix: update JSON URL to point to the dev branch in configure_model function --- mineru/cli/models_download.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/cli/models_download.py b/mineru/cli/models_download.py index 2bb8922f..d01b335e 100644 --- a/mineru/cli/models_download.py +++ b/mineru/cli/models_download.py @@ -43,7 +43,7 @@ def download_and_modify_json(url, local_filename, modifications): def configure_model(model_dir, model_type): """配置模型""" - json_url = 'https://gcore.jsdelivr.net/gh/opendatalab/MinerU@master/mineru.template.json' + json_url = 'https://gcore.jsdelivr.net/gh/opendatalab/MinerU@dev/mineru.template.json' config_file_name = os.getenv('MINERU_TOOLS_CONFIG_JSON', 'mineru.json') home_dir = os.path.expanduser('~') config_file = os.path.join(home_dir, config_file_name) From b614bef035e9acccbdced9f9c3e16da681cbb6ac Mon Sep 17 00:00:00 2001 From: myhloli Date: Fri, 31 Oct 2025 17:50:59 +0800 Subject: [PATCH 03/28] feat: add multiprocessing support for PDF to image conversion with timeout handling --- mineru/utils/pdf_image_tools.py | 44 ++++++++++++++++++++++++++++++++- 1 file changed, 43 insertions(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index adbb58dd..ca51c994 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -11,6 +11,8 @@ from mineru.utils.pdf_reader import image_to_b64str, image_to_bytes, page_to_ima from .enum_class import ImageType from .hash_utils import str_sha256 +from concurrent.futures import ProcessPoolExecutor, TimeoutError as FuturesTimeoutError + def pdf_page_to_image(page: pdfium.PdfPage, dpi=200, image_type=ImageType.PIL) -> dict: """Convert pdfium.PdfDocument to image, Then convert the image to base64. @@ -34,8 +36,48 @@ def pdf_page_to_image(page: pdfium.PdfPage, dpi=200, image_type=ImageType.PIL) - return image_dict +#@todo 通过多进程渲染pdf提速 +def _load_images_from_pdf_worker(pdf_bytes, dpi, start_page_id, end_page_id, image_type): + """用于进程池的包装函数""" + return load_images_from_pdf_core(pdf_bytes, dpi, start_page_id, end_page_id, image_type) + def load_images_from_pdf( + pdf_bytes: bytes, + dpi=200, + start_page_id=0, + end_page_id=None, + image_type=ImageType.PIL, + timeout=300, +): + """带超时控制的 PDF 转图片函数 + + Args: + timeout (int): 超时时间(秒),默认 300 秒 + + Raises: + TimeoutError: 当转换超时时抛出 + """ + with ProcessPoolExecutor(max_workers=1) as executor: + future = executor.submit( + _load_images_from_pdf_worker, + pdf_bytes, + dpi, + start_page_id, + end_page_id, + image_type + ) + try: + images_list = future.result(timeout=timeout) + pdf_doc = pdfium.PdfDocument(pdf_bytes) + return images_list, pdf_doc + except FuturesTimeoutError: + logger.error(f"PDF conversion timeout after {timeout}s") + executor.shutdown(wait=False, cancel_futures=True) + raise TimeoutError(f"PDF to images conversion timeout after {timeout}s") + + +def load_images_from_pdf_core( pdf_bytes: bytes, dpi=200, start_page_id=0, @@ -56,7 +98,7 @@ def load_images_from_pdf( image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) images_list.append(image_dict) - return images_list, pdf_doc + return images_list def cut_image(bbox: tuple, page_num: int, page_pil_img, return_path, image_writer: FileBasedDataWriter, scale=2): From 305e3a61e8a4b824e2f56813cb62a8bc33ed7771 Mon Sep 17 00:00:00 2001 From: myhloli Date: Sat, 1 Nov 2025 02:00:04 +0800 Subject: [PATCH 04/28] fix: disable tokenizers parallelism to prevent potential issues --- mineru/cli/common.py | 1 + 1 file changed, 1 insertion(+) diff --git a/mineru/cli/common.py b/mineru/cli/common.py index 86c25fcb..ed63c2ac 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -20,6 +20,7 @@ from mineru.backend.vlm.vlm_analyze import aio_doc_analyze as aio_vlm_doc_analyz pdf_suffixes = ["pdf"] image_suffixes = ["png", "jpeg", "jp2", "webp", "gif", "bmp", "jpg", "tiff"] +os.environ["TOKENIZERS_PARALLELISM"] = "false" def read_fn(path): if not isinstance(path, Path): From 66d5f3dfd2fd75e482a0fe71aff590eb3d9b5f57 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 15:08:31 +0800 Subject: [PATCH 05/28] feat: refactor PDF image conversion to use get_end_page_id utility function and add multi-threading support --- mineru/cli/common.py | 7 +-- mineru/utils/pdf_image_tools.py | 82 +++++++++++++++++++++++++-------- mineru/utils/pdf_page_id.py | 10 ++++ 3 files changed, 75 insertions(+), 24 deletions(-) create mode 100644 mineru/utils/pdf_page_id.py diff --git a/mineru/cli/common.py b/mineru/cli/common.py index ed63c2ac..9dcedc64 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -16,6 +16,7 @@ from mineru.utils.pdf_image_tools import images_bytes_to_pdf_bytes from mineru.backend.vlm.vlm_middle_json_mkcontent import union_make as vlm_union_make from mineru.backend.vlm.vlm_analyze import doc_analyze as vlm_doc_analyze from mineru.backend.vlm.vlm_analyze import aio_doc_analyze as aio_vlm_doc_analyze +from mineru.utils.pdf_page_id import get_end_page_id pdf_suffixes = ["pdf"] image_suffixes = ["png", "jpeg", "jp2", "webp", "gif", "bmp", "jpg", "tiff"] @@ -49,11 +50,7 @@ def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page # 从字节数据加载PDF pdf = pdfium.PdfDocument(pdf_bytes) - # 确定结束页 - end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else len(pdf) - 1 - if end_page_id > len(pdf) - 1: - logger.warning("end_page_id is out of range, use pdf_docs length") - end_page_id = len(pdf) - 1 + end_page_id = get_end_page_id(end_page_id, len(pdf)) # 创建一个新的PDF文档 output_pdf = pdfium.PdfDocument.new() diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index ca51c994..417e0206 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -8,8 +8,9 @@ from PIL import Image from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.pdf_reader import image_to_b64str, image_to_bytes, page_to_image -from .enum_class import ImageType -from .hash_utils import str_sha256 +from mineru.utils.enum_class import ImageType +from mineru.utils.hash_utils import str_sha256 +from mineru.utils.pdf_page_id import get_end_page_id from concurrent.futures import ProcessPoolExecutor, TimeoutError as FuturesTimeoutError @@ -36,7 +37,7 @@ def pdf_page_to_image(page: pdfium.PdfPage, dpi=200, image_type=ImageType.PIL) - return image_dict -#@todo 通过多进程渲染pdf提速 + def _load_images_from_pdf_worker(pdf_bytes, dpi, start_page_id, end_page_id, image_type): """用于进程池的包装函数""" return load_images_from_pdf_core(pdf_bytes, dpi, start_page_id, end_page_id, image_type) @@ -49,30 +50,74 @@ def load_images_from_pdf( end_page_id=None, image_type=ImageType.PIL, timeout=300, + threads=4, ): - """带超时控制的 PDF 转图片函数 + """带超时控制的 PDF 转图片函数,支持多进程加速 Args: timeout (int): 超时时间(秒),默认 300 秒 + threads (int): 进程数,默认 4 Raises: TimeoutError: 当转换超时时抛出 """ - with ProcessPoolExecutor(max_workers=1) as executor: - future = executor.submit( - _load_images_from_pdf_worker, - pdf_bytes, - dpi, - start_page_id, - end_page_id, - image_type - ) + # 根据进程数调整超时时间 + timeout = timeout // threads + + pdf_doc = pdfium.PdfDocument(pdf_bytes) + end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) + + # 计算总页数 + total_pages = end_page_id - start_page_id + 1 + + # 实际使用的进程数不超过总页数 + actual_threads = min(threads, total_pages) + + # 根据实际进程数分组页面范围 + pages_per_thread = max(1, total_pages // actual_threads) + page_ranges = [] + + for i in range(actual_threads): + range_start = start_page_id + i * pages_per_thread + if i == actual_threads - 1: + # 最后一个线程处理剩余所有页面 + range_end = end_page_id + else: + range_end = min(start_page_id + (i + 1) * pages_per_thread - 1, end_page_id) + + page_ranges.append((range_start, range_end)) + + with ProcessPoolExecutor(max_workers=actual_threads) as executor: + # 提交所有任务 + futures = [] + for range_start, range_end in page_ranges: + future = executor.submit( + _load_images_from_pdf_worker, + pdf_bytes, + dpi, + range_start, + range_end, + image_type + ) + futures.append((range_start, future)) + try: - images_list = future.result(timeout=timeout) - pdf_doc = pdfium.PdfDocument(pdf_bytes) + # 收集结果并按页码排序 + all_results = [] + for range_start, future in futures: + images_list = future.result(timeout=timeout) + all_results.append((range_start, images_list)) + + # 按起始页码排序并合并结果 + all_results.sort(key=lambda x: x[0]) + images_list = [] + for _, imgs in all_results: + images_list.extend(imgs) + return images_list, pdf_doc except FuturesTimeoutError: logger.error(f"PDF conversion timeout after {timeout}s") + pdf_doc.close() executor.shutdown(wait=False, cancel_futures=True) raise TimeoutError(f"PDF to images conversion timeout after {timeout}s") @@ -87,10 +132,7 @@ def load_images_from_pdf_core( images_list = [] pdf_doc = pdfium.PdfDocument(pdf_bytes) pdf_page_num = len(pdf_doc) - end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else pdf_page_num - 1 - if end_page_id > pdf_page_num - 1: - logger.warning("end_page_id is out of range, use images length") - end_page_id = pdf_page_num - 1 + end_page_id = get_end_page_id(end_page_id, pdf_page_num) for index in range(0, pdf_page_num): if start_page_id <= index <= end_page_id: @@ -98,6 +140,8 @@ def load_images_from_pdf_core( image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) images_list.append(image_dict) + pdf_doc.close() + return images_list diff --git a/mineru/utils/pdf_page_id.py b/mineru/utils/pdf_page_id.py new file mode 100644 index 00000000..3ad88993 --- /dev/null +++ b/mineru/utils/pdf_page_id.py @@ -0,0 +1,10 @@ +# Copyright (c) Opendatalab. All rights reserved. +from loguru import logger + + +def get_end_page_id(end_page_id, pdf_page_num): + end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else pdf_page_num - 1 + if end_page_id > pdf_page_num - 1: + logger.warning("end_page_id is out of range, use images length") + end_page_id = pdf_page_num - 1 + return end_page_id \ No newline at end of file From 05e114f8b9312559e5afcc5e5784991bf13139eb Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 15:32:40 +0800 Subject: [PATCH 06/28] feat: implement multiprocessing for PDF conversion to enhance performance --- mineru/cli/common.py | 18 ++++++++++++++---- 1 file changed, 14 insertions(+), 4 deletions(-) diff --git a/mineru/cli/common.py b/mineru/cli/common.py index 9dcedc64..84d71e4d 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -3,6 +3,7 @@ import io import json import os import copy +from multiprocessing import Pool from pathlib import Path import pypdfium2 as pdfium @@ -77,12 +78,21 @@ def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page return output_bytes +def _convert_pdf_in_process(args): + """在独立进程中执行PDF转换""" + pdf_bytes, start_page_id, end_page_id = args + return convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) + + def _prepare_pdf_bytes(pdf_bytes_list, start_page_id, end_page_id): """准备处理PDF字节数据""" - result = [] - for pdf_bytes in pdf_bytes_list: - new_pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) - result.append(new_pdf_bytes) + # 准备参数列表 + args_list = [(pdf_bytes, start_page_id, end_page_id) for pdf_bytes in pdf_bytes_list] + + # 使用进程池执行转换 + with Pool(processes=min(len(pdf_bytes_list), min(os.cpu_count() or 1, 4))) as pool: + result = pool.map(_convert_pdf_in_process, args_list) + return result From bffc6aff5387bdb9796de7ec2e94efc06a375397 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 15:44:33 +0800 Subject: [PATCH 07/28] fix: streamline PDF conversion process by restructuring try-except block and ensuring proper resource management --- mineru/cli/common.py | 13 ++++--------- 1 file changed, 4 insertions(+), 9 deletions(-) diff --git a/mineru/cli/common.py b/mineru/cli/common.py index 84d71e4d..ec84a4a2 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -47,15 +47,11 @@ def prepare_env(output_dir, pdf_file_name, parse_method): def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page_id=None): + pdf = pdfium.PdfDocument(pdf_bytes) + output_pdf = pdfium.PdfDocument.new() try: - # 从字节数据加载PDF - pdf = pdfium.PdfDocument(pdf_bytes) - end_page_id = get_end_page_id(end_page_id, len(pdf)) - # 创建一个新的PDF文档 - output_pdf = pdfium.PdfDocument.new() - # 选择要导入的页面索引 page_indices = list(range(start_page_id, end_page_id + 1)) @@ -68,13 +64,12 @@ def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page # 获取字节数据 output_bytes = output_buffer.getvalue() - - pdf.close() # 关闭原PDF文档以释放资源 - output_pdf.close() # 关闭新PDF文档以释放资源 except Exception as e: logger.warning(f"Error in converting PDF bytes: {e}, Using original PDF bytes.") output_bytes = pdf_bytes + pdf.close() + output_pdf.close() return output_bytes From 4214634de8d258a82d313da8d097d0da6e2e6f25 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 18:48:31 +0800 Subject: [PATCH 08/28] feat: add timing logs for PDF byte preparation to improve performance monitoring --- mineru/cli/common.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/mineru/cli/common.py b/mineru/cli/common.py index ec84a4a2..eef7622d 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -3,6 +3,7 @@ import io import json import os import copy +import time from multiprocessing import Pool from pathlib import Path @@ -81,6 +82,7 @@ def _convert_pdf_in_process(args): def _prepare_pdf_bytes(pdf_bytes_list, start_page_id, end_page_id): """准备处理PDF字节数据""" + start_time = time.time() # 准备参数列表 args_list = [(pdf_bytes, start_page_id, end_page_id) for pdf_bytes in pdf_bytes_list] @@ -88,6 +90,15 @@ def _prepare_pdf_bytes(pdf_bytes_list, start_page_id, end_page_id): with Pool(processes=min(len(pdf_bytes_list), min(os.cpu_count() or 1, 4))) as pool: result = pool.map(_convert_pdf_in_process, args_list) + logger.debug(f"Prepare PDF bytes cost: {round(time.time() - start_time, 2)}s") + + start_time = time.time() + result = [] + for pdf_bytes in pdf_bytes_list: + new_pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) + result.append(new_pdf_bytes) + logger.debug(f"Prepare PDF bytes cost: {round(time.time() - start_time, 2)}s") + return result From c32ff88400fdbf259560e03f14b0e11cf070aada Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 19:07:19 +0800 Subject: [PATCH 09/28] refactor: rename check_mac_env to check_sys_env and add Windows environment detection --- mineru/cli/client.py | 2 +- mineru/cli/common.py | 26 +++++++++---------- mineru/cli/gradio_app.py | 2 +- .../{check_mac_env.py => check_sys_env.py} | 4 +++ 4 files changed, 19 insertions(+), 15 deletions(-) rename mineru/utils/{check_mac_env.py => check_sys_env.py} (92%) diff --git a/mineru/cli/client.py b/mineru/cli/client.py index fae1e888..a6b127e9 100644 --- a/mineru/cli/client.py +++ b/mineru/cli/client.py @@ -4,7 +4,7 @@ import click from pathlib import Path from loguru import logger -from mineru.utils.check_mac_env import is_mac_os_version_supported +from mineru.utils.check_sys_env import is_mac_os_version_supported from mineru.utils.cli_parser import arg_parse from mineru.utils.config_reader import get_device from mineru.utils.guess_suffix_or_lang import guess_suffix_by_path diff --git a/mineru/cli/common.py b/mineru/cli/common.py index eef7622d..78297aa3 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -11,6 +11,7 @@ import pypdfium2 as pdfium from loguru import logger from mineru.data.data_reader_writer import FileBasedDataWriter +from mineru.utils.check_sys_env import is_windows_environment from mineru.utils.draw_bbox import draw_layout_bbox, draw_span_bbox, draw_line_sort_bbox from mineru.utils.enum_class import MakeMode from mineru.utils.guess_suffix_or_lang import guess_suffix_by_bytes @@ -82,21 +83,20 @@ def _convert_pdf_in_process(args): def _prepare_pdf_bytes(pdf_bytes_list, start_page_id, end_page_id): """准备处理PDF字节数据""" - start_time = time.time() - # 准备参数列表 - args_list = [(pdf_bytes, start_page_id, end_page_id) for pdf_bytes in pdf_bytes_list] - - # 使用进程池执行转换 - with Pool(processes=min(len(pdf_bytes_list), min(os.cpu_count() or 1, 4))) as pool: - result = pool.map(_convert_pdf_in_process, args_list) - - logger.debug(f"Prepare PDF bytes cost: {round(time.time() - start_time, 2)}s") - start_time = time.time() result = [] - for pdf_bytes in pdf_bytes_list: - new_pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) - result.append(new_pdf_bytes) + if is_windows_environment(): + for pdf_bytes in pdf_bytes_list: + new_pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) + result.append(new_pdf_bytes) + else: + # 准备参数列表 + args_list = [(pdf_bytes, start_page_id, end_page_id) for pdf_bytes in pdf_bytes_list] + + # 使用进程池执行转换 + with Pool(processes=min(len(pdf_bytes_list), min(os.cpu_count() or 1, 4))) as pool: + result = pool.map(_convert_pdf_in_process, args_list) + logger.debug(f"Prepare PDF bytes cost: {round(time.time() - start_time, 2)}s") return result diff --git a/mineru/cli/gradio_app.py b/mineru/cli/gradio_app.py index 4157d9d7..b29bf987 100644 --- a/mineru/cli/gradio_app.py +++ b/mineru/cli/gradio_app.py @@ -13,7 +13,7 @@ from gradio_pdf import PDF from loguru import logger from mineru.cli.common import prepare_env, read_fn, aio_do_parse, pdf_suffixes, image_suffixes -from mineru.utils.check_mac_env import is_mac_os_version_supported +from mineru.utils.check_sys_env import is_mac_os_version_supported from mineru.utils.cli_parser import arg_parse from mineru.utils.hash_utils import str_sha256 diff --git a/mineru/utils/check_mac_env.py b/mineru/utils/check_sys_env.py similarity index 92% rename from mineru/utils/check_mac_env.py rename to mineru/utils/check_sys_env.py index 41129c47..31fbfc34 100644 --- a/mineru/utils/check_mac_env.py +++ b/mineru/utils/check_sys_env.py @@ -4,6 +4,10 @@ import platform from packaging import version +def is_windows_environment() -> bool: + return platform.system() == "Windows" + + # Detect if the current environment is a Mac computer def is_mac_environment() -> bool: return platform.system() == "Darwin" From 4afa045545d0f2c4cfc72ec89adf100200de40a4 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 19:10:25 +0800 Subject: [PATCH 10/28] refactor: update import statement to use check_sys_env and adjust logging level for image loading --- mineru/backend/vlm/vlm_analyze.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mineru/backend/vlm/vlm_analyze.py b/mineru/backend/vlm/vlm_analyze.py index ad481753..a317f69e 100644 --- a/mineru/backend/vlm/vlm_analyze.py +++ b/mineru/backend/vlm/vlm_analyze.py @@ -8,7 +8,7 @@ from .utils import enable_custom_logits_processors, set_default_gpu_memory_utili from .model_output_to_middle_json import result_to_middle_json from ...data.data_reader_writer import DataWriter from mineru.utils.pdf_image_tools import load_images_from_pdf -from ...utils.check_mac_env import is_mac_os_version_supported +from ...utils.check_sys_env import is_mac_os_version_supported from ...utils.config_reader import get_device from ...utils.enum_class import ImageType @@ -177,7 +177,7 @@ async def aio_doc_analyze( images_list, pdf_doc = load_images_from_pdf(pdf_bytes, image_type=ImageType.PIL) images_pil_list = [image_dict["img_pil"] for image_dict in images_list] # load_images_time = round(time.time() - load_images_start, 2) - # logger.info(f"load images cost: {load_images_time}, speed: {round(len(images_base64_list)/load_images_time, 3)} images/s") + # logger.debug(f"load images cost: {load_images_time}, speed: {round(len(images_pil_list)/load_images_time, 3)} images/s") # infer_start = time.time() results = await predictor.aio_batch_two_step_extract(images=images_pil_list) From 245ae28c2723f3a0e3e55378c83c76954f094015 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 19:11:05 +0800 Subject: [PATCH 11/28] refactor: optimize page range calculation and enhance logging for image conversion process --- mineru/backend/pipeline/pipeline_analyze.py | 3 +++ mineru/utils/pdf_image_tools.py | 14 ++++++++------ 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/mineru/backend/pipeline/pipeline_analyze.py b/mineru/backend/pipeline/pipeline_analyze.py index 246d0111..e57da57b 100644 --- a/mineru/backend/pipeline/pipeline_analyze.py +++ b/mineru/backend/pipeline/pipeline_analyze.py @@ -99,7 +99,10 @@ def doc_analyze( _lang = lang_list[pdf_idx] # 收集每个数据集中的页面 + # load_images_start = time.time() images_list, pdf_doc = load_images_from_pdf(pdf_bytes, image_type=ImageType.PIL) + # load_images_time = round(time.time() - load_images_start, 2) + # logger.debug(f"load images cost: {load_images_time}, speed: {round(len(images_list) / load_images_time, 3)} images/s") all_image_lists.append(images_list) all_pdf_docs.append(pdf_doc) for page_idx in range(len(images_list)): diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 417e0206..865fa400 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -83,10 +83,12 @@ def load_images_from_pdf( # 最后一个线程处理剩余所有页面 range_end = end_page_id else: - range_end = min(start_page_id + (i + 1) * pages_per_thread - 1, end_page_id) + range_end = start_page_id + (i + 1) * pages_per_thread - 1 page_ranges.append((range_start, range_end)) + # logger.debug(f"PDF to images using {actual_threads} processes, page ranges: {page_ranges}") + with ProcessPoolExecutor(max_workers=actual_threads) as executor: # 提交所有任务 futures = [] @@ -134,11 +136,11 @@ def load_images_from_pdf_core( pdf_page_num = len(pdf_doc) end_page_id = get_end_page_id(end_page_id, pdf_page_num) - for index in range(0, pdf_page_num): - if start_page_id <= index <= end_page_id: - page = pdf_doc[index] - image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) - images_list.append(image_dict) + for index in range(start_page_id, end_page_id+1): + # logger.debug(f"Converting page {index}/{pdf_page_num} to image") + page = pdf_doc[index] + image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) + images_list.append(image_dict) pdf_doc.close() From 5999f6664f09d8f9d17d10e8930d51769ec88190 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 19:31:39 +0800 Subject: [PATCH 12/28] refactor: simplify PDF byte preparation by removing multiprocessing and enhancing direct conversion --- mineru/cli/common.py | 24 +++--------------------- 1 file changed, 3 insertions(+), 21 deletions(-) diff --git a/mineru/cli/common.py b/mineru/cli/common.py index 78297aa3..219f2a78 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -75,30 +75,12 @@ def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page return output_bytes -def _convert_pdf_in_process(args): - """在独立进程中执行PDF转换""" - pdf_bytes, start_page_id, end_page_id = args - return convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) - - def _prepare_pdf_bytes(pdf_bytes_list, start_page_id, end_page_id): """准备处理PDF字节数据""" - start_time = time.time() result = [] - if is_windows_environment(): - for pdf_bytes in pdf_bytes_list: - new_pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) - result.append(new_pdf_bytes) - else: - # 准备参数列表 - args_list = [(pdf_bytes, start_page_id, end_page_id) for pdf_bytes in pdf_bytes_list] - - # 使用进程池执行转换 - with Pool(processes=min(len(pdf_bytes_list), min(os.cpu_count() or 1, 4))) as pool: - result = pool.map(_convert_pdf_in_process, args_list) - - logger.debug(f"Prepare PDF bytes cost: {round(time.time() - start_time, 2)}s") - + for pdf_bytes in pdf_bytes_list: + new_pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) + result.append(new_pdf_bytes) return result From 5349fd7ccdcf1f3ba4d3aaf62d7864c5ac2695b8 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 19:41:22 +0800 Subject: [PATCH 13/28] refactor: enhance PDF image loading by removing multiprocessing for Windows environment and improving logging --- mineru/backend/pipeline/pipeline_analyze.py | 6 +- mineru/utils/pdf_image_tools.py | 111 +++++++++++--------- 2 files changed, 64 insertions(+), 53 deletions(-) diff --git a/mineru/backend/pipeline/pipeline_analyze.py b/mineru/backend/pipeline/pipeline_analyze.py index e57da57b..aa78007f 100644 --- a/mineru/backend/pipeline/pipeline_analyze.py +++ b/mineru/backend/pipeline/pipeline_analyze.py @@ -99,10 +99,10 @@ def doc_analyze( _lang = lang_list[pdf_idx] # 收集每个数据集中的页面 - # load_images_start = time.time() + load_images_start = time.time() images_list, pdf_doc = load_images_from_pdf(pdf_bytes, image_type=ImageType.PIL) - # load_images_time = round(time.time() - load_images_start, 2) - # logger.debug(f"load images cost: {load_images_time}, speed: {round(len(images_list) / load_images_time, 3)} images/s") + load_images_time = round(time.time() - load_images_start, 2) + logger.debug(f"load images cost: {load_images_time}, speed: {round(len(images_list) / load_images_time, 3)} images/s") all_image_lists.append(images_list) all_pdf_docs.append(pdf_doc) for page_idx in range(len(images_list)): diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 865fa400..32e0f831 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -7,6 +7,7 @@ from loguru import logger from PIL import Image from mineru.data.data_reader_writer import FileBasedDataWriter +from mineru.utils.check_sys_env import is_windows_environment from mineru.utils.pdf_reader import image_to_b64str, image_to_bytes, page_to_image from mineru.utils.enum_class import ImageType from mineru.utils.hash_utils import str_sha256 @@ -61,67 +62,77 @@ def load_images_from_pdf( Raises: TimeoutError: 当转换超时时抛出 """ - # 根据进程数调整超时时间 - timeout = timeout // threads - pdf_doc = pdfium.PdfDocument(pdf_bytes) - end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) + if is_windows_environment(): + # Windows 环境下不使用多进程 + return load_images_from_pdf_core( + pdf_bytes, + dpi, + start_page_id, + end_page_id, + image_type + ), pdf_doc + else: + # 根据进程数调整超时时间 + timeout = timeout // threads - # 计算总页数 - total_pages = end_page_id - start_page_id + 1 + end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) - # 实际使用的进程数不超过总页数 - actual_threads = min(threads, total_pages) + # 计算总页数 + total_pages = end_page_id - start_page_id + 1 - # 根据实际进程数分组页面范围 - pages_per_thread = max(1, total_pages // actual_threads) - page_ranges = [] + # 实际使用的进程数不超过总页数 + actual_threads = min(threads, total_pages) - for i in range(actual_threads): - range_start = start_page_id + i * pages_per_thread - if i == actual_threads - 1: - # 最后一个线程处理剩余所有页面 - range_end = end_page_id - else: - range_end = start_page_id + (i + 1) * pages_per_thread - 1 + # 根据实际进程数分组页面范围 + pages_per_thread = max(1, total_pages // actual_threads) + page_ranges = [] - page_ranges.append((range_start, range_end)) + for i in range(actual_threads): + range_start = start_page_id + i * pages_per_thread + if i == actual_threads - 1: + # 最后一个线程处理剩余所有页面 + range_end = end_page_id + else: + range_end = start_page_id + (i + 1) * pages_per_thread - 1 - # logger.debug(f"PDF to images using {actual_threads} processes, page ranges: {page_ranges}") + page_ranges.append((range_start, range_end)) - with ProcessPoolExecutor(max_workers=actual_threads) as executor: - # 提交所有任务 - futures = [] - for range_start, range_end in page_ranges: - future = executor.submit( - _load_images_from_pdf_worker, - pdf_bytes, - dpi, - range_start, - range_end, - image_type - ) - futures.append((range_start, future)) + # logger.debug(f"PDF to images using {actual_threads} processes, page ranges: {page_ranges}") - try: - # 收集结果并按页码排序 - all_results = [] - for range_start, future in futures: - images_list = future.result(timeout=timeout) - all_results.append((range_start, images_list)) + with ProcessPoolExecutor(max_workers=actual_threads) as executor: + # 提交所有任务 + futures = [] + for range_start, range_end in page_ranges: + future = executor.submit( + _load_images_from_pdf_worker, + pdf_bytes, + dpi, + range_start, + range_end, + image_type + ) + futures.append((range_start, future)) - # 按起始页码排序并合并结果 - all_results.sort(key=lambda x: x[0]) - images_list = [] - for _, imgs in all_results: - images_list.extend(imgs) + try: + # 收集结果并按页码排序 + all_results = [] + for range_start, future in futures: + images_list = future.result(timeout=timeout) + all_results.append((range_start, images_list)) - return images_list, pdf_doc - except FuturesTimeoutError: - logger.error(f"PDF conversion timeout after {timeout}s") - pdf_doc.close() - executor.shutdown(wait=False, cancel_futures=True) - raise TimeoutError(f"PDF to images conversion timeout after {timeout}s") + # 按起始页码排序并合并结果 + all_results.sort(key=lambda x: x[0]) + images_list = [] + for _, imgs in all_results: + images_list.extend(imgs) + + return images_list, pdf_doc + except FuturesTimeoutError: + logger.error(f"PDF conversion timeout after {timeout}s") + pdf_doc.close() + executor.shutdown(wait=False, cancel_futures=True) + raise TimeoutError(f"PDF to images conversion timeout after {timeout}s") def load_images_from_pdf_core( From ace7f768696c1de7dcc61399f579357c49fef935 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 20:26:34 +0800 Subject: [PATCH 14/28] refactor: move PDF byte conversion functions to pdf_page_tools and simplify logic --- demo/demo.py | 3 ++- mineru/cli/common.py | 34 +--------------------------- mineru/utils/pdf_image_tools.py | 16 +++++++------ mineru/utils/pdf_page_id.py | 10 --------- mineru/utils/pdf_page_tools.py | 40 +++++++++++++++++++++++++++++++++ tests/unittest/test_e2e.py | 2 +- 6 files changed, 53 insertions(+), 52 deletions(-) delete mode 100644 mineru/utils/pdf_page_id.py create mode 100644 mineru/utils/pdf_page_tools.py diff --git a/demo/demo.py b/demo/demo.py index 5da0dbf1..f2e258da 100644 --- a/demo/demo.py +++ b/demo/demo.py @@ -6,7 +6,8 @@ from pathlib import Path from loguru import logger -from mineru.cli.common import convert_pdf_bytes_to_bytes_by_pypdfium2, prepare_env, read_fn +from mineru.cli.common import prepare_env, read_fn +from mineru.utils.pdf_page_tools import convert_pdf_bytes_to_bytes_by_pypdfium2 from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.draw_bbox import draw_layout_bbox, draw_span_bbox from mineru.utils.enum_class import MakeMode diff --git a/mineru/cli/common.py b/mineru/cli/common.py index 219f2a78..399eac79 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -1,17 +1,12 @@ # Copyright (c) Opendatalab. All rights reserved. -import io import json import os import copy -import time -from multiprocessing import Pool from pathlib import Path -import pypdfium2 as pdfium from loguru import logger from mineru.data.data_reader_writer import FileBasedDataWriter -from mineru.utils.check_sys_env import is_windows_environment from mineru.utils.draw_bbox import draw_layout_bbox, draw_span_bbox, draw_line_sort_bbox from mineru.utils.enum_class import MakeMode from mineru.utils.guess_suffix_or_lang import guess_suffix_by_bytes @@ -19,7 +14,7 @@ from mineru.utils.pdf_image_tools import images_bytes_to_pdf_bytes from mineru.backend.vlm.vlm_middle_json_mkcontent import union_make as vlm_union_make from mineru.backend.vlm.vlm_analyze import doc_analyze as vlm_doc_analyze from mineru.backend.vlm.vlm_analyze import aio_doc_analyze as aio_vlm_doc_analyze -from mineru.utils.pdf_page_id import get_end_page_id +from mineru.utils.pdf_page_tools import convert_pdf_bytes_to_bytes_by_pypdfium2 pdf_suffixes = ["pdf"] image_suffixes = ["png", "jpeg", "jp2", "webp", "gif", "bmp", "jpg", "tiff"] @@ -48,33 +43,6 @@ def prepare_env(output_dir, pdf_file_name, parse_method): return local_image_dir, local_md_dir -def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page_id=None): - pdf = pdfium.PdfDocument(pdf_bytes) - output_pdf = pdfium.PdfDocument.new() - try: - end_page_id = get_end_page_id(end_page_id, len(pdf)) - - # 选择要导入的页面索引 - page_indices = list(range(start_page_id, end_page_id + 1)) - - # 从原PDF导入页面到新PDF - output_pdf.import_pages(pdf, page_indices) - - # 将新PDF保存到内存缓冲区 - output_buffer = io.BytesIO() - output_pdf.save(output_buffer) - - # 获取字节数据 - output_bytes = output_buffer.getvalue() - except Exception as e: - logger.warning(f"Error in converting PDF bytes: {e}, Using original PDF bytes.") - output_bytes = pdf_bytes - - pdf.close() - output_pdf.close() - return output_bytes - - def _prepare_pdf_bytes(pdf_bytes_list, start_page_id, end_page_id): """准备处理PDF字节数据""" result = [] diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 32e0f831..65d6c3f9 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -11,7 +11,7 @@ from mineru.utils.check_sys_env import is_windows_environment from mineru.utils.pdf_reader import image_to_b64str, image_to_bytes, page_to_image from mineru.utils.enum_class import ImageType from mineru.utils.hash_utils import str_sha256 -from mineru.utils.pdf_page_id import get_end_page_id +from mineru.utils.pdf_page_tools import get_end_page_id, convert_pdf_bytes_to_bytes_by_pypdfium2 from concurrent.futures import ProcessPoolExecutor, TimeoutError as FuturesTimeoutError @@ -69,7 +69,7 @@ def load_images_from_pdf( pdf_bytes, dpi, start_page_id, - end_page_id, + get_end_page_id(end_page_id, len(pdf_doc)), image_type ), pdf_doc else: @@ -143,13 +143,15 @@ def load_images_from_pdf_core( image_type=ImageType.PIL, # PIL or BASE64 ): images_list = [] + pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) pdf_doc = pdfium.PdfDocument(pdf_bytes) - pdf_page_num = len(pdf_doc) - end_page_id = get_end_page_id(end_page_id, pdf_page_num) + # pdf_page_num = len(pdf_doc) + # end_page_id = get_end_page_id(end_page_id, pdf_page_num) - for index in range(start_page_id, end_page_id+1): - # logger.debug(f"Converting page {index}/{pdf_page_num} to image") - page = pdf_doc[index] + # for index in range(start_page_id, end_page_id+1): + # # logger.debug(f"Converting page {index}/{pdf_page_num} to image") + # page = pdf_doc[index] + for page in pdf_doc: image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) images_list.append(image_dict) diff --git a/mineru/utils/pdf_page_id.py b/mineru/utils/pdf_page_id.py deleted file mode 100644 index 3ad88993..00000000 --- a/mineru/utils/pdf_page_id.py +++ /dev/null @@ -1,10 +0,0 @@ -# Copyright (c) Opendatalab. All rights reserved. -from loguru import logger - - -def get_end_page_id(end_page_id, pdf_page_num): - end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else pdf_page_num - 1 - if end_page_id > pdf_page_num - 1: - logger.warning("end_page_id is out of range, use images length") - end_page_id = pdf_page_num - 1 - return end_page_id \ No newline at end of file diff --git a/mineru/utils/pdf_page_tools.py b/mineru/utils/pdf_page_tools.py new file mode 100644 index 00000000..9ea957a5 --- /dev/null +++ b/mineru/utils/pdf_page_tools.py @@ -0,0 +1,40 @@ +# Copyright (c) Opendatalab. All rights reserved. +import io + +import pypdfium2 as pdfium +from loguru import logger + + +def get_end_page_id(end_page_id, pdf_page_num): + end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else pdf_page_num - 1 + if end_page_id > pdf_page_num - 1: + logger.warning("end_page_id is out of range, use images length") + end_page_id = pdf_page_num - 1 + return end_page_id + + +def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page_id=None): + pdf = pdfium.PdfDocument(pdf_bytes) + output_pdf = pdfium.PdfDocument.new() + try: + end_page_id = get_end_page_id(end_page_id, len(pdf)) + + # 选择要导入的页面索引 + page_indices = list(range(start_page_id, end_page_id + 1)) + + # 从原PDF导入页面到新PDF + output_pdf.import_pages(pdf, page_indices) + + # 将新PDF保存到内存缓冲区 + output_buffer = io.BytesIO() + output_pdf.save(output_buffer) + + # 获取字节数据 + output_bytes = output_buffer.getvalue() + except Exception as e: + logger.warning(f"Error in converting PDF bytes: {e}, Using original PDF bytes.") + output_bytes = pdf_bytes + + pdf.close() + output_pdf.close() + return output_bytes diff --git a/tests/unittest/test_e2e.py b/tests/unittest/test_e2e.py index d50e69a2..d018f9dc 100644 --- a/tests/unittest/test_e2e.py +++ b/tests/unittest/test_e2e.py @@ -7,10 +7,10 @@ from loguru import logger from bs4 import BeautifulSoup from fuzzywuzzy import fuzz from mineru.cli.common import ( - convert_pdf_bytes_to_bytes_by_pypdfium2, prepare_env, read_fn, ) +from mineru.utils.pdf_page_tools import convert_pdf_bytes_to_bytes_by_pypdfium2 from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.enum_class import MakeMode from mineru.backend.vlm.vlm_analyze import doc_analyze as vlm_doc_analyze From b4c57116c1da22acb5e0c2c141e11253d4c7a645 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 20:57:18 +0800 Subject: [PATCH 15/28] refactor: move PDF byte conversion logic to pdf_page_id and simplify image conversion process --- demo/demo.py | 3 +-- mineru/cli/common.py | 31 ++++++++++++++++++++++++- mineru/utils/pdf_image_tools.py | 14 +++++------- mineru/utils/pdf_page_id.py | 10 +++++++++ mineru/utils/pdf_page_tools.py | 40 --------------------------------- tests/unittest/test_e2e.py | 3 +-- 6 files changed, 48 insertions(+), 53 deletions(-) create mode 100644 mineru/utils/pdf_page_id.py delete mode 100644 mineru/utils/pdf_page_tools.py diff --git a/demo/demo.py b/demo/demo.py index f2e258da..c52018a6 100644 --- a/demo/demo.py +++ b/demo/demo.py @@ -6,8 +6,7 @@ from pathlib import Path from loguru import logger -from mineru.cli.common import prepare_env, read_fn -from mineru.utils.pdf_page_tools import convert_pdf_bytes_to_bytes_by_pypdfium2 +from mineru.cli.common import prepare_env, read_fn, convert_pdf_bytes_to_bytes_by_pypdfium2 from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.draw_bbox import draw_layout_bbox, draw_span_bbox from mineru.utils.enum_class import MakeMode diff --git a/mineru/cli/common.py b/mineru/cli/common.py index 399eac79..445003f2 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -1,10 +1,12 @@ # Copyright (c) Opendatalab. All rights reserved. +import io import json import os import copy from pathlib import Path from loguru import logger +import pypdfium2 as pdfium from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.draw_bbox import draw_layout_bbox, draw_span_bbox, draw_line_sort_bbox @@ -14,7 +16,7 @@ from mineru.utils.pdf_image_tools import images_bytes_to_pdf_bytes from mineru.backend.vlm.vlm_middle_json_mkcontent import union_make as vlm_union_make from mineru.backend.vlm.vlm_analyze import doc_analyze as vlm_doc_analyze from mineru.backend.vlm.vlm_analyze import aio_doc_analyze as aio_vlm_doc_analyze -from mineru.utils.pdf_page_tools import convert_pdf_bytes_to_bytes_by_pypdfium2 +from mineru.utils.pdf_page_id import get_end_page_id pdf_suffixes = ["pdf"] image_suffixes = ["png", "jpeg", "jp2", "webp", "gif", "bmp", "jpg", "tiff"] @@ -43,6 +45,33 @@ def prepare_env(output_dir, pdf_file_name, parse_method): return local_image_dir, local_md_dir +def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page_id=None): + pdf = pdfium.PdfDocument(pdf_bytes) + output_pdf = pdfium.PdfDocument.new() + try: + end_page_id = get_end_page_id(end_page_id, len(pdf)) + + # 选择要导入的页面索引 + page_indices = list(range(start_page_id, end_page_id + 1)) + + # 从原PDF导入页面到新PDF + output_pdf.import_pages(pdf, page_indices) + + # 将新PDF保存到内存缓冲区 + output_buffer = io.BytesIO() + output_pdf.save(output_buffer) + + # 获取字节数据 + output_bytes = output_buffer.getvalue() + except Exception as e: + logger.warning(f"Error in converting PDF bytes: {e}, Using original PDF bytes.") + output_bytes = pdf_bytes + + pdf.close() + output_pdf.close() + return output_bytes + + def _prepare_pdf_bytes(pdf_bytes_list, start_page_id, end_page_id): """准备处理PDF字节数据""" result = [] diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 65d6c3f9..f59877f9 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -11,7 +11,7 @@ from mineru.utils.check_sys_env import is_windows_environment from mineru.utils.pdf_reader import image_to_b64str, image_to_bytes, page_to_image from mineru.utils.enum_class import ImageType from mineru.utils.hash_utils import str_sha256 -from mineru.utils.pdf_page_tools import get_end_page_id, convert_pdf_bytes_to_bytes_by_pypdfium2 +from mineru.utils.pdf_page_id import get_end_page_id from concurrent.futures import ProcessPoolExecutor, TimeoutError as FuturesTimeoutError @@ -143,15 +143,13 @@ def load_images_from_pdf_core( image_type=ImageType.PIL, # PIL or BASE64 ): images_list = [] - pdf_bytes = convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id, end_page_id) pdf_doc = pdfium.PdfDocument(pdf_bytes) - # pdf_page_num = len(pdf_doc) - # end_page_id = get_end_page_id(end_page_id, pdf_page_num) + pdf_page_num = len(pdf_doc) + end_page_id = get_end_page_id(end_page_id, pdf_page_num) - # for index in range(start_page_id, end_page_id+1): - # # logger.debug(f"Converting page {index}/{pdf_page_num} to image") - # page = pdf_doc[index] - for page in pdf_doc: + for index in range(start_page_id, end_page_id+1): + # logger.debug(f"Converting page {index}/{pdf_page_num} to image") + page = pdf_doc[index] image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) images_list.append(image_dict) diff --git a/mineru/utils/pdf_page_id.py b/mineru/utils/pdf_page_id.py new file mode 100644 index 00000000..c1e72336 --- /dev/null +++ b/mineru/utils/pdf_page_id.py @@ -0,0 +1,10 @@ +# Copyright (c) Opendatalab. All rights reserved. +from loguru import logger + + +def get_end_page_id(end_page_id, pdf_page_num): + end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else pdf_page_num - 1 + if end_page_id > pdf_page_num - 1: + logger.warning("end_page_id is out of range, use images length") + end_page_id = pdf_page_num - 1 + return end_page_id diff --git a/mineru/utils/pdf_page_tools.py b/mineru/utils/pdf_page_tools.py deleted file mode 100644 index 9ea957a5..00000000 --- a/mineru/utils/pdf_page_tools.py +++ /dev/null @@ -1,40 +0,0 @@ -# Copyright (c) Opendatalab. All rights reserved. -import io - -import pypdfium2 as pdfium -from loguru import logger - - -def get_end_page_id(end_page_id, pdf_page_num): - end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else pdf_page_num - 1 - if end_page_id > pdf_page_num - 1: - logger.warning("end_page_id is out of range, use images length") - end_page_id = pdf_page_num - 1 - return end_page_id - - -def convert_pdf_bytes_to_bytes_by_pypdfium2(pdf_bytes, start_page_id=0, end_page_id=None): - pdf = pdfium.PdfDocument(pdf_bytes) - output_pdf = pdfium.PdfDocument.new() - try: - end_page_id = get_end_page_id(end_page_id, len(pdf)) - - # 选择要导入的页面索引 - page_indices = list(range(start_page_id, end_page_id + 1)) - - # 从原PDF导入页面到新PDF - output_pdf.import_pages(pdf, page_indices) - - # 将新PDF保存到内存缓冲区 - output_buffer = io.BytesIO() - output_pdf.save(output_buffer) - - # 获取字节数据 - output_bytes = output_buffer.getvalue() - except Exception as e: - logger.warning(f"Error in converting PDF bytes: {e}, Using original PDF bytes.") - output_bytes = pdf_bytes - - pdf.close() - output_pdf.close() - return output_bytes diff --git a/tests/unittest/test_e2e.py b/tests/unittest/test_e2e.py index d018f9dc..6457508f 100644 --- a/tests/unittest/test_e2e.py +++ b/tests/unittest/test_e2e.py @@ -8,9 +8,8 @@ from bs4 import BeautifulSoup from fuzzywuzzy import fuzz from mineru.cli.common import ( prepare_env, - read_fn, + read_fn, convert_pdf_bytes_to_bytes_by_pypdfium2, ) -from mineru.utils.pdf_page_tools import convert_pdf_bytes_to_bytes_by_pypdfium2 from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.enum_class import MakeMode from mineru.backend.vlm.vlm_analyze import doc_analyze as vlm_doc_analyze From 20793957746a86338b3b8ccccac932726693ca6c Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 21:02:58 +0800 Subject: [PATCH 16/28] refactor: adjust thread count based on CPU cores and comment out image loading time logging --- mineru/backend/pipeline/pipeline_analyze.py | 6 +++--- mineru/utils/pdf_image_tools.py | 2 ++ 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/mineru/backend/pipeline/pipeline_analyze.py b/mineru/backend/pipeline/pipeline_analyze.py index aa78007f..e57da57b 100644 --- a/mineru/backend/pipeline/pipeline_analyze.py +++ b/mineru/backend/pipeline/pipeline_analyze.py @@ -99,10 +99,10 @@ def doc_analyze( _lang = lang_list[pdf_idx] # 收集每个数据集中的页面 - load_images_start = time.time() + # load_images_start = time.time() images_list, pdf_doc = load_images_from_pdf(pdf_bytes, image_type=ImageType.PIL) - load_images_time = round(time.time() - load_images_start, 2) - logger.debug(f"load images cost: {load_images_time}, speed: {round(len(images_list) / load_images_time, 3)} images/s") + # load_images_time = round(time.time() - load_images_start, 2) + # logger.debug(f"load images cost: {load_images_time}, speed: {round(len(images_list) / load_images_time, 3)} images/s") all_image_lists.append(images_list) all_pdf_docs.append(pdf_doc) for page_idx in range(len(images_list)): diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index f59877f9..15e74128 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -1,4 +1,5 @@ # Copyright (c) Opendatalab. All rights reserved. +import os from io import BytesIO import numpy as np @@ -74,6 +75,7 @@ def load_images_from_pdf( ), pdf_doc else: # 根据进程数调整超时时间 + threads = min(os.cpu_count(), threads) timeout = timeout // threads end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) From 2f120db20e8f7358dba62616181338c2d22bf95d Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 21:24:13 +0800 Subject: [PATCH 17/28] fix: update JSON URL to point to the master branch for model configuration --- mineru/cli/models_download.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/cli/models_download.py b/mineru/cli/models_download.py index d01b335e..2bb8922f 100644 --- a/mineru/cli/models_download.py +++ b/mineru/cli/models_download.py @@ -43,7 +43,7 @@ def download_and_modify_json(url, local_filename, modifications): def configure_model(model_dir, model_type): """配置模型""" - json_url = 'https://gcore.jsdelivr.net/gh/opendatalab/MinerU@dev/mineru.template.json' + json_url = 'https://gcore.jsdelivr.net/gh/opendatalab/MinerU@master/mineru.template.json' config_file_name = os.getenv('MINERU_TOOLS_CONFIG_JSON', 'mineru.json') home_dir = os.path.expanduser('~') config_file = os.path.join(home_dir, config_file_name) From 54417a51f86f6a4792a743247c1fe9a3e48f2c4f Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 3 Nov 2025 21:27:00 +0800 Subject: [PATCH 18/28] refactor: reorder import statements for clarity and consistency --- demo/demo.py | 2 +- tests/unittest/test_e2e.py | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/demo/demo.py b/demo/demo.py index c52018a6..5da0dbf1 100644 --- a/demo/demo.py +++ b/demo/demo.py @@ -6,7 +6,7 @@ from pathlib import Path from loguru import logger -from mineru.cli.common import prepare_env, read_fn, convert_pdf_bytes_to_bytes_by_pypdfium2 +from mineru.cli.common import convert_pdf_bytes_to_bytes_by_pypdfium2, prepare_env, read_fn from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.draw_bbox import draw_layout_bbox, draw_span_bbox from mineru.utils.enum_class import MakeMode diff --git a/tests/unittest/test_e2e.py b/tests/unittest/test_e2e.py index 6457508f..d50e69a2 100644 --- a/tests/unittest/test_e2e.py +++ b/tests/unittest/test_e2e.py @@ -7,8 +7,9 @@ from loguru import logger from bs4 import BeautifulSoup from fuzzywuzzy import fuzz from mineru.cli.common import ( + convert_pdf_bytes_to_bytes_by_pypdfium2, prepare_env, - read_fn, convert_pdf_bytes_to_bytes_by_pypdfium2, + read_fn, ) from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.enum_class import MakeMode From 6250c453d92f27434d302a0a1879f076c8142392 Mon Sep 17 00:00:00 2001 From: Xiaomeng Zhao Date: Mon, 3 Nov 2025 22:04:49 +0800 Subject: [PATCH 19/28] Update mineru/utils/pdf_image_tools.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --- mineru/utils/pdf_image_tools.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 15e74128..a7203d2d 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -149,7 +149,7 @@ def load_images_from_pdf_core( pdf_page_num = len(pdf_doc) end_page_id = get_end_page_id(end_page_id, pdf_page_num) - for index in range(start_page_id, end_page_id+1): + for index in range(start_page_id, end_page_id + 1): # logger.debug(f"Converting page {index}/{pdf_page_num} to image") page = pdf_doc[index] image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) From 74de2725cb6ea22e87fa5aca68e6907a64fcad13 Mon Sep 17 00:00:00 2001 From: Xiaomeng Zhao Date: Mon, 3 Nov 2025 22:08:20 +0800 Subject: [PATCH 20/28] Update mineru/utils/pdf_image_tools.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --- mineru/utils/pdf_image_tools.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index a7203d2d..fb06db80 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -75,7 +75,7 @@ def load_images_from_pdf( ), pdf_doc else: # 根据进程数调整超时时间 - threads = min(os.cpu_count(), threads) + threads = min(os.cpu_count() or 1, threads) timeout = timeout // threads end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) From a9c9501af6c1901ded3d6c12863fee98b02254c9 Mon Sep 17 00:00:00 2001 From: Xiaomeng Zhao Date: Mon, 3 Nov 2025 22:09:29 +0800 Subject: [PATCH 21/28] Update mineru/utils/pdf_image_tools.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --- mineru/utils/pdf_image_tools.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index fb06db80..c98e3134 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -76,8 +76,6 @@ def load_images_from_pdf( else: # 根据进程数调整超时时间 threads = min(os.cpu_count() or 1, threads) - timeout = timeout // threads - end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) # 计算总页数 From 51df4d850891570737c805a945dd0e3e1dce4a93 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 4 Nov 2025 09:54:45 +0800 Subject: [PATCH 22/28] refactor: enhance PDF conversion function parameters and improve thread handling logic --- mineru/utils/pdf_image_tools.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index c98e3134..5bbaeadb 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -57,6 +57,11 @@ def load_images_from_pdf( """带超时控制的 PDF 转图片函数,支持多进程加速 Args: + pdf_bytes (bytes): PDF 文件的 bytes + dpi (int, optional): reset the dpi of dpi. Defaults to 200. + start_page_id (int, optional): 起始页码. Defaults to 0. + end_page_id (int | None, optional): 结束页码. Defaults to None. + image_type (ImageType, optional): 图片类型. Defaults to ImageType.PIL. timeout (int): 超时时间(秒),默认 300 秒 threads (int): 进程数,默认 4 @@ -74,15 +79,13 @@ def load_images_from_pdf( image_type ), pdf_doc else: - # 根据进程数调整超时时间 - threads = min(os.cpu_count() or 1, threads) end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) # 计算总页数 total_pages = end_page_id - start_page_id + 1 # 实际使用的进程数不超过总页数 - actual_threads = min(threads, total_pages) + actual_threads = min(min(os.cpu_count() or 1, threads), total_pages) # 根据实际进程数分组页面范围 pages_per_thread = max(1, total_pages // actual_threads) @@ -129,7 +132,6 @@ def load_images_from_pdf( return images_list, pdf_doc except FuturesTimeoutError: - logger.error(f"PDF conversion timeout after {timeout}s") pdf_doc.close() executor.shutdown(wait=False, cancel_futures=True) raise TimeoutError(f"PDF to images conversion timeout after {timeout}s") From be2369bdd417b2a799b89dc9175f2bc15dff287d Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 4 Nov 2025 19:09:33 +0800 Subject: [PATCH 23/28] feat: add ONNX configuration for thread management and integrate into table structure --- .../table/rec/slanet_plus/table_structure.py | 4 ++++ .../rec/unet_table/table_structure_unet.py | 5 ++++ mineru/model/utils/onnx_config.py | 24 +++++++++++++++++++ 3 files changed, 33 insertions(+) create mode 100644 mineru/model/utils/onnx_config.py diff --git a/mineru/model/table/rec/slanet_plus/table_structure.py b/mineru/model/table/rec/slanet_plus/table_structure.py index c36c61db..99c704bc 100644 --- a/mineru/model/table/rec/slanet_plus/table_structure.py +++ b/mineru/model/table/rec/slanet_plus/table_structure.py @@ -16,6 +16,7 @@ from typing import Any, Dict, List, Tuple import numpy as np +from mineru.model.utils.onnx_config import get_op_num_threads from .table_structure_utils import ( OrtInferSession, TableLabelDecode, @@ -29,6 +30,9 @@ class TableStructurer: self.preprocess_op = TablePreprocess() self.batch_preprocess_op = BatchTablePreprocess() + config["intra_op_num_threads"] = get_op_num_threads("MINERU_INTRA_OP_NUM_THREADS") + config["inter_op_num_threads"] = get_op_num_threads("MINERU_INTER_OP_NUM_THREADS") + self.session = OrtInferSession(config) self.character = self.session.get_metadata() diff --git a/mineru/model/table/rec/unet_table/table_structure_unet.py b/mineru/model/table/rec/unet_table/table_structure_unet.py index 9c6b9826..642dee2d 100644 --- a/mineru/model/table/rec/unet_table/table_structure_unet.py +++ b/mineru/model/table/rec/unet_table/table_structure_unet.py @@ -5,6 +5,8 @@ from typing import Optional, Dict, Any, Tuple import cv2 import numpy as np from skimage import measure + +from mineru.model.utils.onnx_config import get_op_num_threads from .utils import OrtInferSession, resize_img from .utils_table_line_rec import ( get_table_line, @@ -28,6 +30,9 @@ class TSRUnet: self.inp_height = 1024 self.inp_width = 1024 + config["intra_op_num_threads"] = get_op_num_threads("MINERU_INTRA_OP_NUM_THREADS") + config["inter_op_num_threads"] = get_op_num_threads("MINERU_INTER_OP_NUM_THREADS") + self.session = OrtInferSession(config) def __call__( diff --git a/mineru/model/utils/onnx_config.py b/mineru/model/utils/onnx_config.py new file mode 100644 index 00000000..492ea743 --- /dev/null +++ b/mineru/model/utils/onnx_config.py @@ -0,0 +1,24 @@ +import os + + +def get_op_num_threads(env_name: str) -> int: + env_value = os.getenv(env_name, None) + return get_op_num_threads_from_value(env_value) + + +def get_op_num_threads_from_value(env_value: str) -> int: + if env_value is not None: + try: + num_threads = int(env_value) + if num_threads > 0: + return num_threads + except ValueError: + return -1 + return -1 + + +if __name__ == '__main__': + print(get_op_num_threads_from_value('1')) + print(get_op_num_threads_from_value('0')) + print(get_op_num_threads_from_value('-1')) + print(get_op_num_threads_from_value('abc')) \ No newline at end of file From 5de8f1a19f06551a4549806a5621ebd5dddaff1f Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 4 Nov 2025 19:47:59 +0800 Subject: [PATCH 24/28] feat: add environment variables for PDF rendering timeout and ONNX thread management --- docs/en/usage/cli_tools.md | 11 +++++++ docs/zh/usage/cli_tools.md | 12 ++++++++ .../table/rec/slanet_plus/table_structure.py | 2 +- .../rec/unet_table/table_structure_unet.py | 2 +- mineru/model/utils/onnx_config.py | 24 --------------- mineru/utils/os_env_config.py | 30 +++++++++++++++++++ mineru/utils/pdf_image_tools.py | 5 +++- 7 files changed, 59 insertions(+), 27 deletions(-) delete mode 100644 mineru/model/utils/onnx_config.py create mode 100644 mineru/utils/os_env_config.py diff --git a/docs/en/usage/cli_tools.md b/docs/en/usage/cli_tools.md index 14d81408..5b81027e 100644 --- a/docs/en/usage/cli_tools.md +++ b/docs/en/usage/cli_tools.md @@ -100,3 +100,14 @@ Here are the environment variables and their descriptions: * Used to enable table merging functionality * Default is `true`, can be set to `false` via environment variable to disable table merging functionality. +- `MINERU_PDF_LOAD_IMAGES_TIMEOUT`: + * Used to set the timeout period (in seconds) for rendering PDF to images + * Default is `300` seconds, can be set to other values via environment variable to adjust the image rendering timeout. + +- `MINERU_INTRA_OP_NUM_THREADS`: + * Used to set the intra_op thread count for ONNX models, affects the computation speed of individual operators + * Default is `-1` (auto-select), can be set to other values via environment variable to adjust the thread count. + +- `MINERU_INTER_OP_NUM_THREADS`: + * Used to set the inter_op thread count for ONNX models, affects the parallel execution of multiple operators + * Default is `-1` (auto-select), can be set to other values via environment variable to adjust the thread count. diff --git a/docs/zh/usage/cli_tools.md b/docs/zh/usage/cli_tools.md index 99f0c370..15d4d2eb 100644 --- a/docs/zh/usage/cli_tools.md +++ b/docs/zh/usage/cli_tools.md @@ -94,3 +94,15 @@ MinerU命令行工具的某些参数存在相同功能的环境变量配置, - `MINERU_TABLE_MERGE_ENABLE`: * 用于启用表格合并功能 * 默认为`true`,可通过环境变量设置为`false`来禁用表格合并功能。 + +- `MINERU_PDF_LOAD_IMAGES_TIMEOUT`: + * 用于设置将PDF渲染为图片的超时时间(秒) + * 默认为`300`秒,可通过环境变量设置为其他值以调整渲染图片的超时时间。 + +- `MINERU_INTRA_OP_NUM_THREADS`: + * 用于设置onnx模型的intra_op线程数,影响单个算子的计算速度 + * 默认为`-1`(自动选择),可通过环境变量设置为其他值以调整线程数。 + +- `MINERU_INTER_OP_NUM_THREADS`: + * 用于设置onnx模型的inter_op线程数,影响多个算子的并行执行 + * 默认为`-1`(自动选择),可通过环境变量设置为其他值以调整线程数。 diff --git a/mineru/model/table/rec/slanet_plus/table_structure.py b/mineru/model/table/rec/slanet_plus/table_structure.py index 99c704bc..d9b2f0ce 100644 --- a/mineru/model/table/rec/slanet_plus/table_structure.py +++ b/mineru/model/table/rec/slanet_plus/table_structure.py @@ -16,7 +16,7 @@ from typing import Any, Dict, List, Tuple import numpy as np -from mineru.model.utils.onnx_config import get_op_num_threads +from mineru.utils.os_env_config import get_op_num_threads from .table_structure_utils import ( OrtInferSession, TableLabelDecode, diff --git a/mineru/model/table/rec/unet_table/table_structure_unet.py b/mineru/model/table/rec/unet_table/table_structure_unet.py index 642dee2d..d20ced76 100644 --- a/mineru/model/table/rec/unet_table/table_structure_unet.py +++ b/mineru/model/table/rec/unet_table/table_structure_unet.py @@ -6,7 +6,7 @@ import cv2 import numpy as np from skimage import measure -from mineru.model.utils.onnx_config import get_op_num_threads +from mineru.utils.os_env_config import get_op_num_threads from .utils import OrtInferSession, resize_img from .utils_table_line_rec import ( get_table_line, diff --git a/mineru/model/utils/onnx_config.py b/mineru/model/utils/onnx_config.py deleted file mode 100644 index 492ea743..00000000 --- a/mineru/model/utils/onnx_config.py +++ /dev/null @@ -1,24 +0,0 @@ -import os - - -def get_op_num_threads(env_name: str) -> int: - env_value = os.getenv(env_name, None) - return get_op_num_threads_from_value(env_value) - - -def get_op_num_threads_from_value(env_value: str) -> int: - if env_value is not None: - try: - num_threads = int(env_value) - if num_threads > 0: - return num_threads - except ValueError: - return -1 - return -1 - - -if __name__ == '__main__': - print(get_op_num_threads_from_value('1')) - print(get_op_num_threads_from_value('0')) - print(get_op_num_threads_from_value('-1')) - print(get_op_num_threads_from_value('abc')) \ No newline at end of file diff --git a/mineru/utils/os_env_config.py b/mineru/utils/os_env_config.py new file mode 100644 index 00000000..684976ca --- /dev/null +++ b/mineru/utils/os_env_config.py @@ -0,0 +1,30 @@ +import os + + +def get_op_num_threads(env_name: str) -> int: + env_value = os.getenv(env_name, None) + return get_value_from_string(env_value, -1) + + +def get_load_images_timeout() -> int: + env_value = os.getenv('MINERU_PDF_LOAD_IMAGES_TIMEOUT', None) + return get_value_from_string(env_value, 300) + + +def get_value_from_string(env_value: str, default_value: int) -> int: + if env_value is not None: + try: + num_threads = int(env_value) + if num_threads > 0: + return num_threads + except ValueError: + return default_value + return default_value + + +if __name__ == '__main__': + print(get_value_from_string('1', -1)) + print(get_value_from_string('0', -1)) + print(get_value_from_string('-1', -1)) + print(get_value_from_string('abc', -1)) + print(get_load_images_timeout()) \ No newline at end of file diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 5bbaeadb..f8709926 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -9,6 +9,7 @@ from PIL import Image from mineru.data.data_reader_writer import FileBasedDataWriter from mineru.utils.check_sys_env import is_windows_environment +from mineru.utils.os_env_config import get_load_images_timeout from mineru.utils.pdf_reader import image_to_b64str, image_to_bytes, page_to_image from mineru.utils.enum_class import ImageType from mineru.utils.hash_utils import str_sha256 @@ -51,7 +52,7 @@ def load_images_from_pdf( start_page_id=0, end_page_id=None, image_type=ImageType.PIL, - timeout=300, + timeout=None, threads=4, ): """带超时控制的 PDF 转图片函数,支持多进程加速 @@ -79,6 +80,8 @@ def load_images_from_pdf( image_type ), pdf_doc else: + if timeout is None: + timeout = get_load_images_timeout() end_page_id = get_end_page_id(end_page_id, len(pdf_doc)) # 计算总页数 From dae2cc8514a1585b0518198749151010aeb6dfd6 Mon Sep 17 00:00:00 2001 From: Xiaomeng Zhao Date: Tue, 4 Nov 2025 19:59:59 +0800 Subject: [PATCH 25/28] Update mineru/utils/pdf_image_tools.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --- mineru/utils/pdf_image_tools.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index f8709926..41d0237f 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -63,7 +63,7 @@ def load_images_from_pdf( start_page_id (int, optional): 起始页码. Defaults to 0. end_page_id (int | None, optional): 结束页码. Defaults to None. image_type (ImageType, optional): 图片类型. Defaults to ImageType.PIL. - timeout (int): 超时时间(秒),默认 300 秒 + timeout (int | None, optional): 超时时间(秒)。如果为 None,则从环境变量 MINERU_PDF_LOAD_IMAGES_TIMEOUT 读取,若未设置则默认为 300 秒。 threads (int): 进程数,默认 4 Raises: From 5ec07ee7aba67401ca257de4e8cdf5c5bbe2a584 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 4 Nov 2025 20:18:14 +0800 Subject: [PATCH 26/28] feat: update environment variable for PDF rendering timeout and enhance documentation --- README.md | 4 ++++ README_zh-CN.md | 4 ++++ docs/en/usage/cli_tools.md | 2 +- docs/zh/usage/cli_tools.md | 2 +- mineru/utils/os_env_config.py | 2 +- 5 files changed, 11 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 6dc6b9a1..83ac47df 100644 --- a/README.md +++ b/README.md @@ -44,6 +44,10 @@ # Changelog +- 2025/11/04 2.6.4 Release + - Added timeout configuration for PDF image rendering, default is 300 seconds, can be configured via environment variable `MINERU_PDF_RENDER_TIMEOUT` to prevent long blocking of the rendering process caused by some abnormal PDF files. + - Added CPU thread count configuration options for ONNX models, default is the system CPU core count, can be configured via environment variables `MINERU_INTRA_OP_NUM_THREADS` and `MINERU_INTER_OP_NUM_THREADS` to reduce CPU resource contention conflicts in high concurrency scenarios. + - 2025/10/31 2.6.3 Release - Added support for a new backend `vlm-mlx-engine`, enabling MLX-accelerated inference for the MinerU2.5 model on Apple Silicon devices. Compared to the `vlm-transformers` backend, `vlm-mlx-engine` delivers a 100%–200% speed improvement. - Bug fixes: #3849, #3859 diff --git a/README_zh-CN.md b/README_zh-CN.md index b36f7332..e9dea28a 100644 --- a/README_zh-CN.md +++ b/README_zh-CN.md @@ -44,6 +44,10 @@ # 更新记录 +- 2025/11/04 2.6.4 发布 + - 为pdf渲染图片增加超时配置,默认为300秒,可通过环境变量`MINERU_PDF_RENDER_TIMEOUT`进行配置,防止部分异常pdf文件导致渲染过程长时间阻塞。 + - 为onnx模型增加cpu线程数配置选项,默认为系统cpu核心数,可通过环境变量`MINERU_INTRA_OP_NUM_THREADS`和`MINERU_INTER_OP_NUM_THREADS`进行配置,以减少高并发场景下的对cpu资源的抢占冲突。 + - 2025/10/31 2.6.3 发布 - 增加新后端`vlm-mlx-engine`支持,在Apple Silicon设备上支持使用`MLX`加速`MinerU2.5`模型推理,相比`vlm-transformers`后端,`vlm-mlx-engine`后端速度提升100%~200%。 - bug修复: #3849 #3859 diff --git a/docs/en/usage/cli_tools.md b/docs/en/usage/cli_tools.md index 5b81027e..ccffbc43 100644 --- a/docs/en/usage/cli_tools.md +++ b/docs/en/usage/cli_tools.md @@ -100,7 +100,7 @@ Here are the environment variables and their descriptions: * Used to enable table merging functionality * Default is `true`, can be set to `false` via environment variable to disable table merging functionality. -- `MINERU_PDF_LOAD_IMAGES_TIMEOUT`: +- `MINERU_PDF_RENDER_TIMEOUT`: * Used to set the timeout period (in seconds) for rendering PDF to images * Default is `300` seconds, can be set to other values via environment variable to adjust the image rendering timeout. diff --git a/docs/zh/usage/cli_tools.md b/docs/zh/usage/cli_tools.md index 15d4d2eb..218da1e2 100644 --- a/docs/zh/usage/cli_tools.md +++ b/docs/zh/usage/cli_tools.md @@ -95,7 +95,7 @@ MinerU命令行工具的某些参数存在相同功能的环境变量配置, * 用于启用表格合并功能 * 默认为`true`,可通过环境变量设置为`false`来禁用表格合并功能。 -- `MINERU_PDF_LOAD_IMAGES_TIMEOUT`: +- `MINERU_PDF_RENDER_TIMEOUT`: * 用于设置将PDF渲染为图片的超时时间(秒) * 默认为`300`秒,可通过环境变量设置为其他值以调整渲染图片的超时时间。 diff --git a/mineru/utils/os_env_config.py b/mineru/utils/os_env_config.py index 684976ca..43d01334 100644 --- a/mineru/utils/os_env_config.py +++ b/mineru/utils/os_env_config.py @@ -7,7 +7,7 @@ def get_op_num_threads(env_name: str) -> int: def get_load_images_timeout() -> int: - env_value = os.getenv('MINERU_PDF_LOAD_IMAGES_TIMEOUT', None) + env_value = os.getenv('MINERU_PDF_RENDER_TIMEOUT', None) return get_value_from_string(env_value, 300) From fe1549960d14951834a748ff721096f5e022f9ba Mon Sep 17 00:00:00 2001 From: Xiaomeng Zhao Date: Tue, 4 Nov 2025 20:20:37 +0800 Subject: [PATCH 27/28] Update mineru/utils/pdf_image_tools.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --- mineru/utils/pdf_image_tools.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 41d0237f..7347b9d5 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -97,7 +97,7 @@ def load_images_from_pdf( for i in range(actual_threads): range_start = start_page_id + i * pages_per_thread if i == actual_threads - 1: - # 最后一个线程处理剩余所有页面 + # 最后一个进程处理剩余所有页面 range_end = end_page_id else: range_end = start_page_id + (i + 1) * pages_per_thread - 1 From e010b0974ac39dbc32fa8586b754d0eac7c22cf5 Mon Sep 17 00:00:00 2001 From: Xiaomeng Zhao Date: Tue, 4 Nov 2025 20:21:37 +0800 Subject: [PATCH 28/28] Update mineru/utils/pdf_image_tools.py Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com> --- mineru/utils/pdf_image_tools.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 7347b9d5..450f024d 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -88,7 +88,7 @@ def load_images_from_pdf( total_pages = end_page_id - start_page_id + 1 # 实际使用的进程数不超过总页数 - actual_threads = min(min(os.cpu_count() or 1, threads), total_pages) + actual_threads = min(os.cpu_count() or 1, threads, total_pages) # 根据实际进程数分组页面范围 pages_per_thread = max(1, total_pages // actual_threads)
解析后端 pipeline
(精度1 82+)
vlm (精度1 90+)vlm (精度1 90+)
transformers