From c9f223d7fa4831c43b96e2e7fda5ba1ebd06a4e3 Mon Sep 17 00:00:00 2001 From: myhloli Date: Fri, 29 May 2026 00:25:39 +0800 Subject: [PATCH 01/21] doc(cli): support boolean client-side output generation option --- docs/en/usage/cli_tools.md | 7 +++++++ docs/zh/usage/cli_tools.md | 6 ++++++ mineru/cli/client.py | 2 +- 3 files changed, 14 insertions(+), 1 deletion(-) diff --git a/docs/en/usage/cli_tools.md b/docs/en/usage/cli_tools.md index 66142060..01d93c1f 100644 --- a/docs/en/usage/cli_tools.md +++ b/docs/en/usage/cli_tools.md @@ -21,6 +21,10 @@ Options: -e, --end INTEGER Ending page number for parsing (0-based) -f, --formula BOOLEAN Enable formula parsing (default: enabled) -t, --table BOOLEAN Enable table parsing (default: enabled) + --client-side-output-generation BOOLEAN + Generate Markdown and content lists locally + from server-returned middle JSON, images, and + original files (default: disabled) --help Show help information ``` > [!TIP] @@ -59,6 +63,9 @@ Options: starts a reusable local mineru-api service. --enable-vlm-preload BOOLEAN Preload the local VLM model when gradio starts a local mineru-api service. + --client-side-output-generation BOOLEAN + Generate Markdown and content lists locally + from server-returned middle JSON. --latex-delimiters-type [a|b|all] Set the type of LaTeX delimiters to use in Markdown rendering: 'a' for type '$', 'b' for diff --git a/docs/zh/usage/cli_tools.md b/docs/zh/usage/cli_tools.md index 489b1476..e99d780b 100644 --- a/docs/zh/usage/cli_tools.md +++ b/docs/zh/usage/cli_tools.md @@ -21,6 +21,9 @@ Options: -e, --end INTEGER 结束解析的页码(从 0 开始) -f, --formula BOOLEAN 是否启用公式解析(默认开启) -t, --table BOOLEAN 是否启用表格解析(默认开启) + --client-side-output-generation BOOLEAN + 在客户端基于服务端返回的 middle JSON、图片与原文件 + 生成 Markdown 和 content list(默认关闭) --help 显示帮助信息 ``` > [!TIP] @@ -54,6 +57,9 @@ Options: mineru-api --enable-vlm-preload BOOLEAN 在 Gradio 拉起本地 mineru-api 时预加载本地 VLM 模型 + --client-side-output-generation BOOLEAN + 在客户端基于服务端返回的 middle JSON 生成 Markdown + 和 content list --latex-delimiters-type [a|b|all] 设置在 Markdown 渲染中使用的 LaTeX 分隔符类型 ('a' 表示 '$' 类型,'b' 表示 '()[]' 类型, diff --git a/mineru/cli/client.py b/mineru/cli/client.py index 0af1f7d6..3ed3a28d 100644 --- a/mineru/cli/client.py +++ b/mineru/cli/client.py @@ -1128,7 +1128,7 @@ async def run_orchestrated_cli( @click.option( "--client-side-output-generation", "client_side_output_generation", - is_flag=True, + type=bool, default=False, help=( "Generate markdown and content lists locally from server-returned " From 2d3fb59c672d3bbeb16a0cfa1e2ac8ded216333e Mon Sep 17 00:00:00 2001 From: myhloli Date: Fri, 29 May 2026 17:22:41 +0800 Subject: [PATCH 02/21] fix: keep model output for client-side output generation --- mineru/cli/api_request.py | 4 ++-- mineru/cli/client.py | 5 ++--- mineru/cli/gradio_app.py | 2 +- 3 files changed, 5 insertions(+), 6 deletions(-) diff --git a/mineru/cli/api_request.py b/mineru/cli/api_request.py index 44f1381b..a90cb8d1 100644 --- a/mineru/cli/api_request.py +++ b/mineru/cli/api_request.py @@ -161,7 +161,7 @@ async def parse_request_form( Form( description=( "Defer final markdown/content-list generation to the client. " - "When enabled, the server returns staged middle JSON and images." + "When enabled, the server returns staged middle JSON, model output, and images." ), ), ] = False, @@ -188,7 +188,7 @@ async def parse_request_form( if client_side_output_generation: return_md = False return_middle_json = True - return_model_output = False + return_model_output = True return_content_list = False return_images = True diff --git a/mineru/cli/client.py b/mineru/cli/client.py index 3ed3a28d..936df82e 100644 --- a/mineru/cli/client.py +++ b/mineru/cli/client.py @@ -626,9 +626,8 @@ def build_request_form_data( image_analysis: bool = True, client_side_output_generation: bool = False, ) -> dict[str, str | list[str]]: - # 开启客户端输出生成时,服务端只负责返回 middle json、图片和原文件。 + # 开启客户端输出生成时,只关闭客户端会重建的最终产物。 return_md = not client_side_output_generation - return_model_output = not client_side_output_generation return_content_list = not client_side_output_generation return _api_client.build_parse_request_form_data( lang_list=[lang], @@ -642,7 +641,7 @@ def build_request_form_data( end_page_id=end_page_id, return_md=return_md, return_middle_json=True, - return_model_output=return_model_output, + return_model_output=True, return_content_list=return_content_list, return_images=True, response_format_zip=True, diff --git a/mineru/cli/gradio_app.py b/mineru/cli/gradio_app.py index 509a3095..e305d69f 100644 --- a/mineru/cli/gradio_app.py +++ b/mineru/cli/gradio_app.py @@ -997,7 +997,7 @@ async def _run_to_markdown_job( end_page_id=end_pages - 1, return_md=not use_client_side_output_generation, return_middle_json=True, - return_model_output=not use_client_side_output_generation, + return_model_output=True, return_content_list=not use_client_side_output_generation, return_images=True, response_format_zip=True, From 05b2414959b378d04fb98231cf2a62132717a64d Mon Sep 17 00:00:00 2001 From: myhloli Date: Fri, 29 May 2026 19:08:58 +0800 Subject: [PATCH 03/21] chore: log title leveling execution duration --- mineru/utils/title_level_postprocess.py | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/mineru/utils/title_level_postprocess.py b/mineru/utils/title_level_postprocess.py index bb7c1066..76a14dd7 100644 --- a/mineru/utils/title_level_postprocess.py +++ b/mineru/utils/title_level_postprocess.py @@ -1,9 +1,12 @@ # Copyright (c) Opendatalab. All rights reserved. from __future__ import annotations +import time from copy import deepcopy from typing import Any +from loguru import logger + from mineru.utils.config_reader import get_llm_aided_config from mineru.utils.llm_aided import llm_aided_title @@ -29,7 +32,15 @@ def _resolve_title_aided_config() -> dict[str, Any] | None: def apply_title_leveling_to_pdf_info(pdf_info: list[dict[str, Any]]): title_aided_config = _resolve_title_aided_config() if title_aided_config: - llm_aided_title(pdf_info, title_aided_config) + start_time = time.perf_counter() + success = False + try: + llm_aided_title(pdf_info, title_aided_config) + success = True + finally: + elapsed = time.perf_counter() - start_time + status = "finished" if success else "failed" + logger.info(f"title leveling {status}, cost: {elapsed:.2f}s") def finalize_client_side_middle_json(middle_json: dict[str, Any]) -> dict[str, Any]: From 11e845b3d0dba91e33cf7abe962b5d7440f1c351 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 1 Jun 2026 10:52:56 +0800 Subject: [PATCH 04/21] fix: improve error handling during PDF render executor shutdown --- mineru/utils/pdf_image_tools.py | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 76fd6e48..cea7a240 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -177,8 +177,14 @@ def _recycle_pdf_render_executor( _pdf_render_executor = None if terminate_processes: - _terminate_executor_processes(executor) - executor.shutdown(wait=False, cancel_futures=True) + try: + _terminate_executor_processes(executor) + except Exception as exc: + logger.warning(f"Failed to terminate PDF render executor processes: {exc}") + try: + executor.shutdown(wait=False, cancel_futures=True) + except Exception as exc: + logger.warning(f"Failed to shutdown PDF render executor: {exc}") def shutdown_pdf_render_executor() -> None: @@ -298,7 +304,9 @@ async def aio_load_images_from_pdf_bytes_range( def _terminate_executor_processes(executor): """强制终止 ProcessPoolExecutor 中的所有子进程""" - processes = list(getattr(executor, "_processes", {}).values()) + # executor.shutdown() 后 _processes 会被置空,重复回收时直接视为无进程。 + process_map = getattr(executor, "_processes", None) or {} + processes = list(process_map.values()) if not processes: return From bd60d158dcb149c56e00c730f460d13b67de3e75 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 1 Jun 2026 10:55:38 +0800 Subject: [PATCH 05/21] fix: update external link to the latest model version --- mineru/resources/gradio_header.html | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/resources/gradio_header.html b/mineru/resources/gradio_header.html index 0d9c69a2..44fb9fd1 100644 --- a/mineru/resources/gradio_header.html +++ b/mineru/resources/gradio_header.html @@ -112,7 +112,7 @@ - + From 02d757fffa9b2e67aef7e0685ba55691b4675542 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 1 Jun 2026 17:30:36 +0800 Subject: [PATCH 06/21] fix: switch PDF rendering to use 'spawn' method for multiprocessing and improve task submission handling --- mineru/cli/client.py | 7 ++++- mineru/utils/pdf_image_tools.py | 50 +++++++++++++++++++++++++++++++-- 2 files changed, 53 insertions(+), 4 deletions(-) diff --git a/mineru/cli/client.py b/mineru/cli/client.py index 936df82e..c00c68d5 100644 --- a/mineru/cli/client.py +++ b/mineru/cli/client.py @@ -1,5 +1,6 @@ # Copyright (c) Opendatalab. All rights reserved. import asyncio +import multiprocessing import os import sys import threading @@ -343,8 +344,12 @@ async def mark_task_completed( def create_visualization_context() -> Optional[VisualizationContext]: try: + spawn_context = multiprocessing.get_context("spawn") return VisualizationContext( - executor=ProcessPoolExecutor(max_workers=1), + executor=ProcessPoolExecutor( + max_workers=1, + mp_context=spawn_context, + ), futures=[], ) except Exception as exc: diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index cea7a240..d06c7a05 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -35,11 +35,15 @@ DEFAULT_PDF_IMAGE_DPI = 200 # DEFAULT_PDF_IMAGE_DPI = 144 MAX_PDF_RENDER_PROCESSES = 3 MIN_PAGES_PER_RENDER_PROCESS = 30 +PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS = 0.1 PDF_RENDER_TERMINATE_GRACE_PERIOD_SECONDS = 0.1 PDF_RENDER_KILL_JOIN_TIMEOUT_SECONDS = 0.1 _pdf_render_executor: ProcessPoolExecutor | None = None _pdf_render_executor_lock = threading.Lock() +_pdf_render_spawn_submit_lock = threading.Lock() +_pdf_render_spawn_submit_executor_id: int | None = None +_pdf_render_spawn_submit_count = 0 def pdf_page_to_image( @@ -137,9 +141,10 @@ def _create_pdf_render_executor(max_workers: int) -> ProcessPoolExecutor: return ProcessPoolExecutor(max_workers=max_workers) start_method = multiprocessing.get_start_method() - if start_method == "fork": + if start_method != "spawn": logger.debug( - "PDF image rendering switches multiprocessing start method from fork to spawn" + "PDF image rendering switches multiprocessing start method " + f"from {start_method} to spawn" ) return ProcessPoolExecutor( max_workers=max_workers, @@ -149,6 +154,44 @@ def _create_pdf_render_executor(max_workers: int) -> ProcessPoolExecutor: return ProcessPoolExecutor(max_workers=max_workers) +def _is_pdf_render_pool_still_spawning_workers(executor: ProcessPoolExecutor) -> bool: + """判断渲染进程池是否还可能因为 submit 而继续创建新的 worker。""" + max_workers = getattr(executor, "_max_workers", None) + if max_workers is None or max_workers <= 1: + return False + + processes = getattr(executor, "_processes", None) + process_count = 0 if processes is None else len(processes) + return process_count < max_workers + + +def _submit_pdf_render_task( + executor: ProcessPoolExecutor, + fn, + *args, + **kwargs, +): + """提交 PDF 渲染任务;冷启动补 worker 时串行 submit 并错开 100ms。""" + global _pdf_render_spawn_submit_executor_id, _pdf_render_spawn_submit_count + + with _pdf_render_spawn_submit_lock: + should_throttle = _is_pdf_render_pool_still_spawning_workers(executor) + if should_throttle: + executor_id = id(executor) + if _pdf_render_spawn_submit_executor_id != executor_id: + _pdf_render_spawn_submit_executor_id = executor_id + _pdf_render_spawn_submit_count = 0 + elif _pdf_render_spawn_submit_count > 0: + time.sleep(PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS) + + future = executor.submit(fn, *args, **kwargs) + + if should_throttle: + _pdf_render_spawn_submit_count += 1 + + return future + + def _get_pdf_render_executor() -> ProcessPoolExecutor: global _pdf_render_executor @@ -238,7 +281,8 @@ def _load_images_from_pdf_bytes_range( futures = [] future_to_range = {} for range_start, range_end in page_ranges: - future = executor.submit( + future = _submit_pdf_render_task( + executor, _load_images_from_pdf_worker, pdf_bytes, dpi, From 7a385e7e94dcba40e04d25081bffab40c1711397 Mon Sep 17 00:00:00 2001 From: myhloli Date: Mon, 1 Jun 2026 20:10:23 +0800 Subject: [PATCH 07/21] fix: enhance CID font usage detection and normalize font name handling in PDF processing --- mineru/utils/pdf_classify.py | 140 +++++++++++++++++++++++++++++++++-- 1 file changed, 135 insertions(+), 5 deletions(-) diff --git a/mineru/utils/pdf_classify.py b/mineru/utils/pdf_classify.py index 08d298fc..854a6b47 100644 --- a/mineru/utils/pdf_classify.py +++ b/mineru/utils/pdf_classify.py @@ -1,5 +1,6 @@ # Copyright (c) Opendatalab. All rights reserved. import re +from ctypes import byref, c_int, create_string_buffer from io import BytesIO import pypdfium2 as pdfium @@ -18,6 +19,8 @@ HIGH_IMAGE_COVERAGE_THRESHOLD = 0.8 TEXT_QUALITY_MIN_CHARS = 300 TEXT_QUALITY_BAD_THRESHOLD = 0.03 UNICODE_MAP_ERROR_RATIO_THRESHOLD = 0.04 +CID_FONT_USAGE_RATIO_THRESHOLD = 0.01 +CID_FONT_USAGE_COUNT_THRESHOLD = 30 MAX_PAGE_ASPECT_RATIO = 10.0 SUSPICIOUS_CJK_72XX_START = 0x7280 SUSPICIOUS_CJK_72XX_END = 0x72DF @@ -103,7 +106,20 @@ def classify(pdf_bytes): ) return "ocr" - if detect_cid_font_signal_pypdf(pdf_bytes, page_indices): + cid_font_signal = get_cid_font_signal_pypdf(pdf_bytes, page_indices) + cid_font_usage_signal = _get_cid_font_usage_signal_from_samples( + text_samples, + cid_font_signal, + ) + if cid_font_usage_signal["triggered"]: + logger.debug( + "Classify PDF as OCR due to high CID font usage without ToUnicode: " + f"page={cid_font_usage_signal['page_index'] + 1}, " + f"fonts={cid_font_usage_signal['font_names']}, " + f"chars={cid_font_usage_signal['cid_font_char_count']}, " + f"total={cid_font_usage_signal['total_chars']}, " + f"ratio={cid_font_usage_signal['cid_font_usage_ratio']:.4f}" + ) return "ocr" text_quality_signal = _get_text_quality_signal_from_samples(text_samples) @@ -321,6 +337,107 @@ def _get_unicode_map_error_signal_from_samples(text_samples): } +def _normalize_pdf_font_name(font_name) -> str: + """规范化 PDF 字体名,统一 pypdf 的 NameObject 和 PDFium 返回值格式。""" + if font_name is None: + return "" + return str(font_name).strip().lstrip("/") + + +def _get_pdfium_char_font_name(text_page, char_index: int) -> str: + """读取 PDFium 字符级字体名,用于统计可疑 CID 字体的实际使用比例。""" + flags = c_int() + buffer_length = pdfium_c.FPDFText_GetFontInfo( + text_page, + char_index, + None, + 0, + byref(flags), + ) + if buffer_length <= 0: + return "" + + font_name_buffer = create_string_buffer(buffer_length) + actual_length = pdfium_c.FPDFText_GetFontInfo( + text_page, + char_index, + font_name_buffer, + buffer_length, + byref(flags), + ) + if actual_length <= 0: + return "" + + return font_name_buffer.value.decode("utf-8", errors="ignore") + + +def _get_cid_font_usage_signal_from_samples(text_samples, cid_font_signal): + """基于 PDFium 字符级字体名统计可疑 CID 字体在抽样页中的真实使用比例。""" + best_signal = { + "triggered": False, + "page_index": None, + "font_names": [], + "cid_font_char_count": 0, + "total_chars": 0, + "cid_font_usage_ratio": 0.0, + } + if not cid_font_signal or not cid_font_signal.get("triggered"): + return best_signal + + page_fonts = cid_font_signal.get("page_fonts") or {} + for text_sample in text_samples: + page_index = text_sample.get("page_index") + cid_font_names = { + _normalize_pdf_font_name(font_name) + for font_name in page_fonts.get(page_index, set()) + } + cid_font_names.discard("") + if not cid_font_names: + continue + + text_page = text_sample["text_page"] + total_chars = text_page.count_chars() + if total_chars <= 0: + continue + + matched_font_names = set() + cid_font_char_count = 0 + for char_index in range(total_chars): + font_name = _normalize_pdf_font_name( + _get_pdfium_char_font_name(text_page, char_index) + ) + if font_name in cid_font_names: + cid_font_char_count += 1 + matched_font_names.add(font_name) + + cid_font_usage_ratio = cid_font_char_count / total_chars + signal = { + "triggered": False, + "page_index": page_index, + "font_names": sorted(matched_font_names), + "cid_font_char_count": cid_font_char_count, + "total_chars": total_chars, + "cid_font_usage_ratio": cid_font_usage_ratio, + } + if ( + cid_font_char_count >= CID_FONT_USAGE_COUNT_THRESHOLD + and cid_font_usage_ratio >= CID_FONT_USAGE_RATIO_THRESHOLD + ): + signal["triggered"] = True + return signal + + if ( + signal["cid_font_usage_ratio"], + signal["cid_font_char_count"], + ) > ( + best_signal["cid_font_usage_ratio"], + best_signal["cid_font_char_count"], + ): + best_signal = signal + + return best_signal + + def _get_u72xx_text_signal_from_samples(text_samples): """基于已缓存的抽样页文本统计扣除常用字后的 U+7280-U+72DF 字符占比。""" cjk_chars = 0 @@ -435,8 +552,10 @@ def _get_sampled_ascii_punct_signal_from_samples(text_samples): return best_signal -def detect_cid_font_signal_pypdf(pdf_bytes, page_indices): +def get_cid_font_signal_pypdf(pdf_bytes, page_indices): + """收集抽样页中无 ToUnicode 的 Identity CID 字体资源,供后续按实际字符使用量判定。""" reader = PdfReader(BytesIO(pdf_bytes)) + page_fonts = {} for page_index in page_indices: page = reader.pages[page_index] @@ -448,7 +567,7 @@ def detect_cid_font_signal_pypdf(pdf_bytes, page_indices): if not fonts: continue - for _, font_ref in fonts.items(): + for font_key, font_ref in fonts.items(): font = _resolve_pdf_object(font_ref) if not font: continue @@ -464,9 +583,20 @@ def detect_cid_font_signal_pypdf(pdf_bytes, page_indices): and has_descendant_fonts and not has_to_unicode ): - return True + font_name = font.get("/BaseFont") or font_key + page_fonts.setdefault(page_index, set()).add( + _normalize_pdf_font_name(font_name) + ) - return False + return { + "triggered": bool(page_fonts), + "page_fonts": page_fonts, + } + + +def detect_cid_font_signal_pypdf(pdf_bytes, page_indices): + """兼容旧接口:只返回是否存在无 ToUnicode 的 Identity CID 字体资源。""" + return get_cid_font_signal_pypdf(pdf_bytes, page_indices)["triggered"] def _resolve_pdf_object(obj): From 8b4c1d3683daa85b7af5739c486c590d2d5fde62 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 00:03:57 +0800 Subject: [PATCH 08/21] fix: refactor inference calls to use dedicated run functions for layout, MFR, and OCR detection/recognition --- mineru/backend/hybrid/hybrid_analyze.py | 34 +++++++++++--- .../hybrid_model_output_to_middle_json.py | 8 +++- mineru/backend/pipeline/batch_analyze.py | 47 ++++++++++++++----- mineru/backend/pipeline/model_init.py | 39 +++++++++++++++ .../pipeline/model_json_to_middle_json.py | 9 +++- 5 files changed, 116 insertions(+), 21 deletions(-) diff --git a/mineru/backend/hybrid/hybrid_analyze.py b/mineru/backend/hybrid/hybrid_analyze.py index da70bbbd..617a1193 100644 --- a/mineru/backend/hybrid/hybrid_analyze.py +++ b/mineru/backend/hybrid/hybrid_analyze.py @@ -19,7 +19,13 @@ from mineru.backend.hybrid.hybrid_model_output_to_middle_json import ( init_middle_json, ) from mineru.backend.utils.runtime_utils import exclude_progress_bar_idle_time -from mineru.backend.pipeline.model_init import HybridModelSingleton +from mineru.backend.pipeline.model_init import ( + HybridModelSingleton, + run_layout_inference, + run_mfr_inference, + run_ocr_det_inference, + run_ocr_rec_inference, +) from mineru.backend.vlm.vlm_analyze import ( ModelSingleton, aio_predictor_execution_guard, @@ -119,8 +125,11 @@ def ocr_det( page_mfd_res, useful_list ) bgr_image = cv2.cvtColor(new_image, cv2.COLOR_RGB2BGR) - ocr_res = hybrid_pipeline_model.ocr_model.ocr( - bgr_image, mfd_res=adjusted_mfdetrec_res, rec=False + ocr_res = run_ocr_det_inference( + hybrid_pipeline_model.ocr_model.ocr, + bgr_image, + mfd_res=adjusted_mfdetrec_res, + rec=False, )[0] if ocr_res: ocr_result_list = get_ocr_result_list( @@ -193,7 +202,11 @@ def ocr_det( # 批处理检测 det_batch_size = min(len(batch_images), batch_ratio * OCR_DET_BASE_BATCH_SIZE) - batch_results = hybrid_pipeline_model.ocr_model.text_detector.batch_predict(batch_images, det_batch_size) + batch_results = run_ocr_det_inference( + hybrid_pipeline_model.ocr_model.text_detector.batch_predict, + batch_images, + det_batch_size, + ) # 处理批处理结果 for crop_info, (dt_boxes, _) in zip(group_crops, batch_results): @@ -392,7 +405,8 @@ def _predict_layout_for_title_split( batch_ratio, ): """执行layout小模型检测,专门为Hybrid标题拆分提供页面layout结果。""" - return hybrid_pipeline_model.layout_model.batch_predict( + return run_layout_inference( + hybrid_pipeline_model.layout_model.batch_predict, images, batch_size=min(8, batch_ratio * LAYOUT_BASE_BATCH_SIZE), ) @@ -433,7 +447,8 @@ def _process_ocr_and_formulas( if inline_formula_enable: images_mfd_res = _build_inline_formula_inputs(images_layout_res) # 公式识别 - inline_formula_list = hybrid_pipeline_model.mfr_model.batch_predict( + inline_formula_list = run_mfr_inference( + hybrid_pipeline_model.mfr_model.batch_predict, images_mfd_res, np_images, batch_size=batch_ratio * MFR_BASE_BATCH_SIZE, @@ -473,7 +488,12 @@ def _process_ocr_and_formulas( img_crop_list.append(ocr_res.pop('np_img')) if len(img_crop_list) > 0: # Process OCR - ocr_result_list = hybrid_pipeline_model.ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0] + ocr_result_list = run_ocr_rec_inference( + hybrid_pipeline_model.ocr_model.ocr, + img_crop_list, + det=False, + tqdm_enable=True, + )[0] # Verify we have matching counts assert len(ocr_result_list) == len(need_ocr_list), f'ocr_result_list: {len(ocr_result_list)}, need_ocr_list: {len(need_ocr_list)}' diff --git a/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py b/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py index 80b75c3f..c8891561 100644 --- a/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py +++ b/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py @@ -14,6 +14,7 @@ from mineru.backend.utils.para_block_utils import ( ) from mineru.backend.hybrid.hybrid_magic_model import MagicModel from mineru.backend.utils.runtime_utils import cross_page_table_merge +from mineru.backend.pipeline.model_init import run_ocr_rec_inference from mineru.utils.config_reader import get_table_enable from mineru.utils.cut_image import cut_image_and_table from mineru.utils.enum_class import ContentType, BlockType @@ -138,7 +139,12 @@ def _apply_post_ocr(pdf_info_list, hybrid_pipeline_model): img_crop_list.append(rotate_vertical_crop_if_needed(span['np_img'])) span.pop('np_img') if len(img_crop_list) > 0: - ocr_res_list = hybrid_pipeline_model.ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0] + ocr_res_list = run_ocr_rec_inference( + hybrid_pipeline_model.ocr_model.ocr, + img_crop_list, + det=False, + tqdm_enable=True, + )[0] assert len(ocr_res_list) == len( need_ocr_list), f'ocr_res_list: {len(ocr_res_list)}, need_ocr_list: {len(need_ocr_list)}' for index, span in enumerate(need_ocr_list): diff --git a/mineru/backend/pipeline/batch_analyze.py b/mineru/backend/pipeline/batch_analyze.py index 82d40a77..78f3f1a0 100644 --- a/mineru/backend/pipeline/batch_analyze.py +++ b/mineru/backend/pipeline/batch_analyze.py @@ -9,7 +9,13 @@ from tqdm import tqdm from collections import defaultdict import numpy as np -from .model_init import AtomModelSingleton +from .model_init import ( + AtomModelSingleton, + run_layout_inference, + run_mfr_inference, + run_ocr_det_inference, + run_ocr_rec_inference, +) from .model_list import AtomicModel from ...utils.config_reader import ( get_formula_enable, @@ -362,9 +368,10 @@ class BatchAnalyze: np_images = [np.asarray(image) for image, _, _ in images_with_extra_info] # pp-doclayout_v2 - images_layout_res += self.model.layout_model.batch_predict( + images_layout_res += run_layout_inference( + self.model.layout_model.batch_predict, pil_images, - batch_size=min(8, self.batch_ratio * LAYOUT_BASE_BATCH_SIZE) + batch_size=min(8, self.batch_ratio * LAYOUT_BASE_BATCH_SIZE), ) # 清理显存 clean_vram(self.model.device, vram_threshold=8) @@ -380,7 +387,8 @@ class BatchAnalyze: images_mfd_res.append(page_formula_res) # 公式识别 - images_formula_list = self.model.mfr_model.batch_predict( + images_formula_list = run_mfr_inference( + self.model.mfr_model.batch_predict, images_mfd_res, np_images, batch_size=self.batch_ratio * MFR_BASE_BATCH_SIZE, @@ -526,7 +534,9 @@ class BatchAnalyze: if inline_mask_boxes else bgr_image ) - ocr_result = det_ocr_engine.ocr(det_image, rec=False)[0] + ocr_result = run_ocr_det_inference( + det_ocr_engine.ocr, det_image, rec=False + )[0] if ocr_result and formula_mask_boxes: ocr_result = update_det_boxes(ocr_result, formula_mask_boxes) if ocr_result: @@ -555,7 +565,13 @@ class BatchAnalyze: enable_merge_det_boxes=False, ) cropped_img_list = [item["cropped_img"] for item in rec_img_list] - ocr_res_list = ocr_engine.ocr(cropped_img_list, det=False, tqdm_enable=True, tqdm_desc=f"Table-ocr rec {_lang}")[0] + ocr_res_list = run_ocr_rec_inference( + ocr_engine.ocr, + cropped_img_list, + det=False, + tqdm_enable=True, + tqdm_desc=f"Table-ocr rec {_lang}", + )[0] # 按照 table_id 将识别结果进行回填 for img_dict, ocr_res in zip(rec_img_list, ocr_res_list): ocr_text = self._normalize_table_ocr_rec_text(ocr_res[0]) @@ -715,7 +731,9 @@ class BatchAnalyze: # 批处理检测 det_batch_size = min(len(batch_images), self.batch_ratio * OCR_DET_BASE_BATCH_SIZE) - batch_results = ocr_model.text_detector.batch_predict(batch_images, det_batch_size) + batch_results = run_ocr_det_inference( + ocr_model.text_detector.batch_predict, batch_images, det_batch_size + ) # 处理批处理结果 for crop_info, (dt_boxes, _) in zip(group_crops, batch_results): @@ -775,8 +793,11 @@ class BatchAnalyze: bgr_image, adjusted_mfdetrec_res, ) - ocr_res = ocr_model.ocr( - det_image, mfd_res=adjusted_mfdetrec_res, rec=False + ocr_res = run_ocr_det_inference( + ocr_model.ocr, + det_image, + mfd_res=adjusted_mfdetrec_res, + rec=False, )[0] # Integration results @@ -831,7 +852,9 @@ class BatchAnalyze: atom_model_name=AtomicModel.OCR, lang=lang ) - ocr_res_list = ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0] + ocr_res_list = run_ocr_rec_inference( + ocr_model.ocr, img_crop_list, det=False, tqdm_enable=True + )[0] # Verify we have matching counts assert len(ocr_res_list) == len( @@ -900,7 +923,9 @@ class BatchAnalyze: ) seal_crop_bgr = cv2.cvtColor(seal_crop_rgb, cv2.COLOR_RGB2BGR) - seal_ocr_res = seal_ocr_model.ocr(seal_crop_bgr, det=True, rec=True)[0] + seal_ocr_res = run_ocr_det_inference( + seal_ocr_model.ocr, seal_crop_bgr, det=True, rec=True + )[0] if not seal_ocr_res: continue diff --git a/mineru/backend/pipeline/model_init.py b/mineru/backend/pipeline/model_init.py index e2c98b66..2006402a 100644 --- a/mineru/backend/pipeline/model_init.py +++ b/mineru/backend/pipeline/model_init.py @@ -19,6 +19,45 @@ from ...utils.enum_class import ModelPath from ...utils.models_download_utils import auto_download_and_get_model_root_path PIPELINE_MODEL_INIT_LOCK = threading.RLock() +# 这些锁保护 pipeline 与 hybrid 共享的 atom model/native 模型推理调用,避免多线程同时进入同一个模型对象。 +PIPELINE_LAYOUT_INFERENCE_LOCK = threading.RLock() +PIPELINE_MFR_INFERENCE_LOCK = threading.RLock() +PIPELINE_OCR_DET_INFERENCE_LOCK = threading.RLock() +PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock() + + +def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs): + """在指定推理锁内执行真实 native 模型调用。""" + with inference_lock: + return inference_callable(*args, **kwargs) + + +def run_layout_inference(inference_callable, *args, **kwargs): + """在共享 Layout 推理锁内执行模型调用。""" + return _run_with_inference_lock( + PIPELINE_LAYOUT_INFERENCE_LOCK, inference_callable, *args, **kwargs + ) + + +def run_mfr_inference(inference_callable, *args, **kwargs): + """在共享 MFR 推理锁内执行模型调用。""" + return _run_with_inference_lock( + PIPELINE_MFR_INFERENCE_LOCK, inference_callable, *args, **kwargs + ) + + +def run_ocr_det_inference(inference_callable, *args, **kwargs): + """在共享 OCR det 推理锁内执行模型调用。""" + return _run_with_inference_lock( + PIPELINE_OCR_DET_INFERENCE_LOCK, inference_callable, *args, **kwargs + ) + + +def run_ocr_rec_inference(inference_callable, *args, **kwargs): + """在共享 OCR rec 推理锁内执行模型调用。""" + return _run_with_inference_lock( + PIPELINE_OCR_REC_INFERENCE_LOCK, inference_callable, *args, **kwargs + ) MFR_MODEL = os.getenv('MINERU_FORMULA_CH_SUPPORT', 'False') if MFR_MODEL.lower() in ['true', '1', 'yes']: diff --git a/mineru/backend/pipeline/model_json_to_middle_json.py b/mineru/backend/pipeline/model_json_to_middle_json.py index 4025af06..d51c5875 100644 --- a/mineru/backend/pipeline/model_json_to_middle_json.py +++ b/mineru/backend/pipeline/model_json_to_middle_json.py @@ -7,7 +7,10 @@ from tqdm import tqdm from mineru.backend.utils.html_image_utils import replace_inline_table_images from mineru.backend.utils.runtime_utils import cross_page_table_merge from mineru.utils.config_reader import get_device -from mineru.backend.pipeline.model_init import AtomModelSingleton +from mineru.backend.pipeline.model_init import ( + AtomModelSingleton, + run_ocr_rec_inference, +) from mineru.backend.pipeline.para_split import para_split from mineru.utils.char_utils import full_to_half from mineru.utils.cut_image import cut_image_and_table @@ -231,7 +234,9 @@ def _apply_post_ocr(pdf_info_list, lang=None): det_db_box_thresh=0.3, lang=lang ) - ocr_res_list = ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0] + ocr_res_list = run_ocr_rec_inference( + ocr_model.ocr, img_crop_list, det=False, tqdm_enable=True + )[0] assert len(ocr_res_list) == len( need_ocr_list), f'ocr_res_list: {len(ocr_res_list)}, need_ocr_list: {len(need_ocr_list)}' for index, span in enumerate(need_ocr_list): From c65e95a443a37ee3cdbb5ff0875434c16c5ded29 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 02:22:55 +0800 Subject: [PATCH 09/21] fix: implement inference admission lock and quiet zone for improved concurrency management --- mineru/backend/pipeline/model_init.py | 30 ++++++++++++++++++++++++++- mineru/utils/model_utils.py | 15 +++++++++++++- 2 files changed, 43 insertions(+), 2 deletions(-) diff --git a/mineru/backend/pipeline/model_init.py b/mineru/backend/pipeline/model_init.py index 2006402a..f696dc00 100644 --- a/mineru/backend/pipeline/model_init.py +++ b/mineru/backend/pipeline/model_init.py @@ -1,6 +1,7 @@ # Copyright (c) Opendatalab. All rights reserved. import os import threading +from contextlib import contextmanager import torch from loguru import logger @@ -24,12 +25,39 @@ PIPELINE_LAYOUT_INFERENCE_LOCK = threading.RLock() PIPELINE_MFR_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_DET_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock() +# 推理准入门闩:清理线程等待静默区时,阻止新的推理调用继续插队进入阶段锁。 +PIPELINE_INFERENCE_ADMISSION_LOCK = threading.RLock() +PIPELINE_INFERENCE_STAGE_LOCKS = ( + PIPELINE_LAYOUT_INFERENCE_LOCK, + PIPELINE_MFR_INFERENCE_LOCK, + PIPELINE_OCR_DET_INFERENCE_LOCK, + PIPELINE_OCR_REC_INFERENCE_LOCK, +) def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs): """在指定推理锁内执行真实 native 模型调用。""" - with inference_lock: + with PIPELINE_INFERENCE_ADMISSION_LOCK: + inference_lock.acquire() + try: return inference_callable(*args, **kwargs) + finally: + inference_lock.release() + + +@contextmanager +def pipeline_inference_quiet_zone(): + """进入 pipeline/hybrid 共享模型推理静默区,等待四类阶段推理全部退出。""" + acquired_locks = [] + with PIPELINE_INFERENCE_ADMISSION_LOCK: + try: + for inference_lock in PIPELINE_INFERENCE_STAGE_LOCKS: + inference_lock.acquire() + acquired_locks.append(inference_lock) + yield + finally: + for inference_lock in reversed(acquired_locks): + inference_lock.release() def run_layout_inference(inference_callable, *args, **kwargs): diff --git a/mineru/utils/model_utils.py b/mineru/utils/model_utils.py index 6c1f486a..1c2e4f68 100644 --- a/mineru/utils/model_utils.py +++ b/mineru/utils/model_utils.py @@ -180,7 +180,15 @@ def get_res_list_from_layout_res(layout_res, overlap_threshold=0.8): return ocr_res_list, table_res_list, single_page_mfdetrec_res -def clean_memory(device='cuda'): +def _pipeline_inference_quiet_zone(): + """懒加载推理静默区,避免 utils 模块导入时和 pipeline backend 形成循环依赖。""" + from mineru.backend.pipeline.model_init import pipeline_inference_quiet_zone + + return pipeline_inference_quiet_zone() + + +def _clean_memory_without_inference(device='cuda'): + """执行实际显存/内存清理动作,调用方负责进入推理静默区。""" if str(device).startswith("cuda"): if torch.cuda.is_available(): torch.cuda.empty_cache() @@ -205,6 +213,11 @@ def clean_memory(device='cuda'): gc.collect() +def clean_memory(device='cuda'): + with _pipeline_inference_quiet_zone(): + _clean_memory_without_inference(device) + + def clean_vram(device, vram_threshold=8): total_memory = get_vram(device) if total_memory and total_memory <= vram_threshold: From fca67ac4a91c65b4f71d2a5d7ad778ce8cf9a30e Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 02:27:38 +0800 Subject: [PATCH 10/21] fix: remove memory cleaning condition for post-processing in middle JSON conversion --- mineru/backend/pipeline/model_json_to_middle_json.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/mineru/backend/pipeline/model_json_to_middle_json.py b/mineru/backend/pipeline/model_json_to_middle_json.py index d51c5875..cff112b1 100644 --- a/mineru/backend/pipeline/model_json_to_middle_json.py +++ b/mineru/backend/pipeline/model_json_to_middle_json.py @@ -285,8 +285,6 @@ def finalize_middle_json( """Apply document-level post processing once all page_info entries are ready.""" apply_server_side_postprocess(pdf_info_list, lang=lang) finalize_middle_json_from_preproc(pdf_info_list) - if os.getenv('MINERU_DONOT_CLEAN_MEM') is None and len(pdf_info_list) >= 10: - clean_memory(get_device()) def init_middle_json(): From aa0bda48beb93a1da79947fdf78d54e2aef22835 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 02:28:09 +0800 Subject: [PATCH 11/21] fix: remove memory cleaning condition for post-processing in middle JSON conversion --- mineru/backend/pipeline/model_json_to_middle_json.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/mineru/backend/pipeline/model_json_to_middle_json.py b/mineru/backend/pipeline/model_json_to_middle_json.py index cff112b1..d9700724 100644 --- a/mineru/backend/pipeline/model_json_to_middle_json.py +++ b/mineru/backend/pipeline/model_json_to_middle_json.py @@ -1,12 +1,10 @@ # Copyright (c) Opendatalab. All rights reserved. import copy -import os from tqdm import tqdm from mineru.backend.utils.html_image_utils import replace_inline_table_images from mineru.backend.utils.runtime_utils import cross_page_table_merge -from mineru.utils.config_reader import get_device from mineru.backend.pipeline.model_init import ( AtomModelSingleton, run_ocr_rec_inference, @@ -16,7 +14,6 @@ from mineru.utils.char_utils import full_to_half from mineru.utils.cut_image import cut_image_and_table from mineru.utils.enum_class import ContentType, BlockType from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info -from mineru.utils.model_utils import clean_memory from mineru.backend.pipeline.pipeline_magic_model import MagicModel from mineru.utils.ocr_utils import OcrConfidence, rotate_vertical_crop_if_needed from mineru.version import __version__ From b068c34dc0ab643b9bff2b7cc7ca570bc37f74e1 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 02:45:19 +0800 Subject: [PATCH 12/21] fix: implement global inference lock for serializing Layout/MFR/OCR processing --- mineru/backend/pipeline/model_init.py | 23 +++++++++++------------ 1 file changed, 11 insertions(+), 12 deletions(-) diff --git a/mineru/backend/pipeline/model_init.py b/mineru/backend/pipeline/model_init.py index f696dc00..27215545 100644 --- a/mineru/backend/pipeline/model_init.py +++ b/mineru/backend/pipeline/model_init.py @@ -25,13 +25,12 @@ PIPELINE_LAYOUT_INFERENCE_LOCK = threading.RLock() PIPELINE_MFR_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_DET_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock() +# 临时全局推理锁:串行化 Layout/MFR/OCR det/OCR rec,验证跨阶段 native 推理并发是否触发 malloc/segfault。 +PIPELINE_GLOBAL_INFERENCE_LOCK = threading.RLock() # 推理准入门闩:清理线程等待静默区时,阻止新的推理调用继续插队进入阶段锁。 PIPELINE_INFERENCE_ADMISSION_LOCK = threading.RLock() PIPELINE_INFERENCE_STAGE_LOCKS = ( - PIPELINE_LAYOUT_INFERENCE_LOCK, - PIPELINE_MFR_INFERENCE_LOCK, - PIPELINE_OCR_DET_INFERENCE_LOCK, - PIPELINE_OCR_REC_INFERENCE_LOCK, + PIPELINE_GLOBAL_INFERENCE_LOCK, ) @@ -61,30 +60,30 @@ def pipeline_inference_quiet_zone(): def run_layout_inference(inference_callable, *args, **kwargs): - """在共享 Layout 推理锁内执行模型调用。""" + """在全局推理锁内执行 Layout 模型调用。""" return _run_with_inference_lock( - PIPELINE_LAYOUT_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_mfr_inference(inference_callable, *args, **kwargs): - """在共享 MFR 推理锁内执行模型调用。""" + """在全局推理锁内执行 MFR 模型调用。""" return _run_with_inference_lock( - PIPELINE_MFR_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_ocr_det_inference(inference_callable, *args, **kwargs): - """在共享 OCR det 推理锁内执行模型调用。""" + """在全局推理锁内执行 OCR det 模型调用。""" return _run_with_inference_lock( - PIPELINE_OCR_DET_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_ocr_rec_inference(inference_callable, *args, **kwargs): - """在共享 OCR rec 推理锁内执行模型调用。""" + """在全局推理锁内执行 OCR rec 模型调用。""" return _run_with_inference_lock( - PIPELINE_OCR_REC_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs ) MFR_MODEL = os.getenv('MINERU_FORMULA_CH_SUPPORT', 'False') From 728e0794e2d6cee7806ab4c1c2c2d1c4bfe96079 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 03:18:18 +0800 Subject: [PATCH 13/21] fix: refactor memory cleaning functions to simplify implementation and avoid circular dependencies --- mineru/utils/model_utils.py | 15 +-------------- 1 file changed, 1 insertion(+), 14 deletions(-) diff --git a/mineru/utils/model_utils.py b/mineru/utils/model_utils.py index 1c2e4f68..6c1f486a 100644 --- a/mineru/utils/model_utils.py +++ b/mineru/utils/model_utils.py @@ -180,15 +180,7 @@ def get_res_list_from_layout_res(layout_res, overlap_threshold=0.8): return ocr_res_list, table_res_list, single_page_mfdetrec_res -def _pipeline_inference_quiet_zone(): - """懒加载推理静默区,避免 utils 模块导入时和 pipeline backend 形成循环依赖。""" - from mineru.backend.pipeline.model_init import pipeline_inference_quiet_zone - - return pipeline_inference_quiet_zone() - - -def _clean_memory_without_inference(device='cuda'): - """执行实际显存/内存清理动作,调用方负责进入推理静默区。""" +def clean_memory(device='cuda'): if str(device).startswith("cuda"): if torch.cuda.is_available(): torch.cuda.empty_cache() @@ -213,11 +205,6 @@ def _clean_memory_without_inference(device='cuda'): gc.collect() -def clean_memory(device='cuda'): - with _pipeline_inference_quiet_zone(): - _clean_memory_without_inference(device) - - def clean_vram(device, vram_threshold=8): total_memory = get_vram(device) if total_memory and total_memory <= vram_threshold: From b169958beb05eecd0f08b0a97133c49934404388 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 03:24:43 +0800 Subject: [PATCH 14/21] fix: reduce maximum PDF render processes to improve resource management --- mineru/utils/pdf_image_tools.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index d06c7a05..80291ca3 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -33,7 +33,7 @@ from concurrent.futures.process import BrokenProcessPool DEFAULT_PDF_IMAGE_DPI = 200 # DEFAULT_PDF_IMAGE_DPI = 144 -MAX_PDF_RENDER_PROCESSES = 3 +MAX_PDF_RENDER_PROCESSES = 1 MIN_PAGES_PER_RENDER_PROCESS = 30 PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS = 0.1 PDF_RENDER_TERMINATE_GRACE_PERIOD_SECONDS = 0.1 From 2d7485547a946f54fc62f6c495cfa3817adc67c2 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 03:26:17 +0800 Subject: [PATCH 15/21] fix: replace global inference lock with specific locks for Layout, MFR, and OCR processing --- mineru/backend/pipeline/model_init.py | 23 ++++++++++++----------- 1 file changed, 12 insertions(+), 11 deletions(-) diff --git a/mineru/backend/pipeline/model_init.py b/mineru/backend/pipeline/model_init.py index 27215545..f696dc00 100644 --- a/mineru/backend/pipeline/model_init.py +++ b/mineru/backend/pipeline/model_init.py @@ -25,12 +25,13 @@ PIPELINE_LAYOUT_INFERENCE_LOCK = threading.RLock() PIPELINE_MFR_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_DET_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock() -# 临时全局推理锁:串行化 Layout/MFR/OCR det/OCR rec,验证跨阶段 native 推理并发是否触发 malloc/segfault。 -PIPELINE_GLOBAL_INFERENCE_LOCK = threading.RLock() # 推理准入门闩:清理线程等待静默区时,阻止新的推理调用继续插队进入阶段锁。 PIPELINE_INFERENCE_ADMISSION_LOCK = threading.RLock() PIPELINE_INFERENCE_STAGE_LOCKS = ( - PIPELINE_GLOBAL_INFERENCE_LOCK, + PIPELINE_LAYOUT_INFERENCE_LOCK, + PIPELINE_MFR_INFERENCE_LOCK, + PIPELINE_OCR_DET_INFERENCE_LOCK, + PIPELINE_OCR_REC_INFERENCE_LOCK, ) @@ -60,30 +61,30 @@ def pipeline_inference_quiet_zone(): def run_layout_inference(inference_callable, *args, **kwargs): - """在全局推理锁内执行 Layout 模型调用。""" + """在共享 Layout 推理锁内执行模型调用。""" return _run_with_inference_lock( - PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_LAYOUT_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_mfr_inference(inference_callable, *args, **kwargs): - """在全局推理锁内执行 MFR 模型调用。""" + """在共享 MFR 推理锁内执行模型调用。""" return _run_with_inference_lock( - PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_MFR_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_ocr_det_inference(inference_callable, *args, **kwargs): - """在全局推理锁内执行 OCR det 模型调用。""" + """在共享 OCR det 推理锁内执行模型调用。""" return _run_with_inference_lock( - PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_OCR_DET_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_ocr_rec_inference(inference_callable, *args, **kwargs): - """在全局推理锁内执行 OCR rec 模型调用。""" + """在共享 OCR rec 推理锁内执行模型调用。""" return _run_with_inference_lock( - PIPELINE_GLOBAL_INFERENCE_LOCK, inference_callable, *args, **kwargs + PIPELINE_OCR_REC_INFERENCE_LOCK, inference_callable, *args, **kwargs ) MFR_MODEL = os.getenv('MINERU_FORMULA_CH_SUPPORT', 'False') From 88d881d29931326e6050abd4d7ce893c7a1801c3 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 15:33:53 +0800 Subject: [PATCH 16/21] fix: increase maximum PDF render processes to enhance rendering performance --- mineru/utils/pdf_image_tools.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index 80291ca3..d06c7a05 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -33,7 +33,7 @@ from concurrent.futures.process import BrokenProcessPool DEFAULT_PDF_IMAGE_DPI = 200 # DEFAULT_PDF_IMAGE_DPI = 144 -MAX_PDF_RENDER_PROCESSES = 1 +MAX_PDF_RENDER_PROCESSES = 3 MIN_PAGES_PER_RENDER_PROCESS = 30 PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS = 0.1 PDF_RENDER_TERMINATE_GRACE_PERIOD_SECONDS = 0.1 From 95a7c5c29050a79124569e714fb8551d610248b5 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 15:51:45 +0800 Subject: [PATCH 17/21] fix: enhance PDF processing by ensuring proper resource management and error handling --- mineru/utils/pdf_classify.py | 191 ++++++++++++++++++++++------------- 1 file changed, 122 insertions(+), 69 deletions(-) diff --git a/mineru/utils/pdf_classify.py b/mineru/utils/pdf_classify.py index 854a6b47..e860eecb 100644 --- a/mineru/utils/pdf_classify.py +++ b/mineru/utils/pdf_classify.py @@ -212,35 +212,97 @@ def get_extreme_aspect_ratio_page_pdfium( page_indices, max_page_aspect_ratio: float = MAX_PAGE_ASPECT_RATIO, ): - for page_index in page_indices: - page = pdf_doc[page_index] - page_width, page_height = page.get_size() - if page_width <= 0 or page_height <= 0: - continue + with pdfium_guard(): + for page_index in page_indices: + page = None + try: + page = pdf_doc[page_index] + page_width, page_height = page.get_size() + if page_width <= 0 or page_height <= 0: + continue - aspect_ratio = max(page_width / page_height, page_height / page_width) - if aspect_ratio > max_page_aspect_ratio: - return page_index, aspect_ratio + aspect_ratio = max(page_width / page_height, page_height / page_width) + if aspect_ratio > max_page_aspect_ratio: + return page_index, aspect_ratio + finally: + _close_pdfium_child(page) return None, None -def _collect_pdfium_text_samples(pdf_doc, page_indices): - """一次性收集抽样页的 textpage 和文本,避免分类链路重复读取 PDFium 文本。""" - text_samples = [] +def _close_pdfium_child(pdfium_obj) -> None: + """显式关闭 PDFium 子对象,避免依赖 weakref/finalizer 延迟释放 native 资源。""" + if pdfium_obj is None: + return + close = getattr(pdfium_obj, "close", None) + if callable(close): + close() - for page_index in page_indices: - page = pdf_doc[page_index] + +def _collect_pdfium_text_sample_from_page(page_index, page): + """从单页 PDFium 对象提取纯 Python 文本统计,并在调用方释放子对象。""" + text_page = None + try: text_page = page.get_textpage() text = text_page.get_text_bounded() - text_samples.append( - { - "page_index": page_index, - "text_page": text_page, - "text": text, - "cleaned_text": re.sub(r"\s+", "", text), - } - ) + char_count = text_page.count_chars() + null_char_count = 0 + replacement_char_count = 0 + control_char_count = 0 + private_use_char_count = 0 + unicode_map_error_count = 0 + font_name_counts = {} + + for char_index in range(char_count): + unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index) + if unicode_code == 0: + null_char_count += 1 + elif unicode_code == 0xFFFD: + replacement_char_count += 1 + elif _is_disallowed_control_unicode(unicode_code): + control_char_count += 1 + elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END: + private_use_char_count += 1 + + if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index): + unicode_map_error_count += 1 + + font_name = _normalize_pdf_font_name( + _get_pdfium_char_font_name(text_page, char_index) + ) + if font_name: + font_name_counts[font_name] = font_name_counts.get(font_name, 0) + 1 + + return { + "page_index": page_index, + "text": text, + "cleaned_text": re.sub(r"\s+", "", text), + "char_count": char_count, + "null_char_count": null_char_count, + "replacement_char_count": replacement_char_count, + "control_char_count": control_char_count, + "private_use_char_count": private_use_char_count, + "unicode_map_error_count": unicode_map_error_count, + "font_name_counts": font_name_counts, + } + finally: + _close_pdfium_child(text_page) + + +def _collect_pdfium_text_samples(pdf_doc, page_indices): + """一次性收集抽样页文本统计,返回纯 Python 数据,避免缓存 PDFium 子对象。""" + text_samples = [] + + with pdfium_guard(): + for page_index in page_indices: + page = None + try: + page = pdf_doc[page_index] + text_samples.append( + _collect_pdfium_text_sample_from_page(page_index, page) + ) + finally: + _close_pdfium_child(page) return text_samples @@ -263,7 +325,7 @@ def get_avg_cleaned_chars_per_page_pdfium(pdf_doc, page_indices): def _get_text_quality_signal_from_samples(text_samples): - """基于已缓存的 PDFium textpage 统计异常字符质量信号。""" + """基于已缓存的抽样页字符计数统计异常字符质量信号。""" total_chars = 0 null_char_count = 0 replacement_char_count = 0 @@ -271,20 +333,11 @@ def _get_text_quality_signal_from_samples(text_samples): private_use_char_count = 0 for text_sample in text_samples: - text_page = text_sample["text_page"] - char_count = text_page.count_chars() - total_chars += char_count - - for char_index in range(char_count): - unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index) - if unicode_code == 0: - null_char_count += 1 - elif unicode_code == 0xFFFD: - replacement_char_count += 1 - elif _is_disallowed_control_unicode(unicode_code): - control_char_count += 1 - elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END: - private_use_char_count += 1 + total_chars += text_sample["char_count"] + null_char_count += text_sample["null_char_count"] + replacement_char_count += text_sample["replacement_char_count"] + control_char_count += text_sample["control_char_count"] + private_use_char_count += text_sample["private_use_char_count"] abnormal_chars = ( null_char_count @@ -318,13 +371,8 @@ def _get_unicode_map_error_signal_from_samples(text_samples): unicode_map_error_count = 0 for text_sample in text_samples: - text_page = text_sample["text_page"] - char_count = text_page.count_chars() - total_chars += char_count - - for char_index in range(char_count): - if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index): - unicode_map_error_count += 1 + total_chars += text_sample["char_count"] + unicode_map_error_count += text_sample["unicode_map_error_count"] unicode_map_error_ratio = 0.0 if total_chars > 0: @@ -395,20 +443,15 @@ def _get_cid_font_usage_signal_from_samples(text_samples, cid_font_signal): if not cid_font_names: continue - text_page = text_sample["text_page"] - total_chars = text_page.count_chars() + total_chars = text_sample["char_count"] if total_chars <= 0: continue - matched_font_names = set() - cid_font_char_count = 0 - for char_index in range(total_chars): - font_name = _normalize_pdf_font_name( - _get_pdfium_char_font_name(text_page, char_index) - ) - if font_name in cid_font_names: - cid_font_char_count += 1 - matched_font_names.add(font_name) + font_name_counts = text_sample.get("font_name_counts") or {} + matched_font_names = cid_font_names.intersection(font_name_counts) + cid_font_char_count = sum( + font_name_counts[font_name] for font_name in matched_font_names + ) cid_font_usage_ratio = cid_font_char_count / total_chars signal = { @@ -608,23 +651,33 @@ def _resolve_pdf_object(obj): def get_high_image_coverage_ratio_pdfium(pdf_doc, page_indices): high_image_coverage_pages = 0 - for page_index in page_indices: - page = pdf_doc[page_index] - page_bbox = page.get_bbox() - page_area = abs( - (page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1]) - ) - image_area = 0.0 + with pdfium_guard(): + for page_index in page_indices: + page = None + try: + page = pdf_doc[page_index] + page_bbox = page.get_bbox() + page_area = abs( + (page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1]) + ) + image_area = 0.0 - for page_object in page.get_objects( - filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3 - ): - left, bottom, right, top = page_object.get_pos() - image_area += max(0.0, right - left) * max(0.0, top - bottom) + for page_object in page.get_objects( + filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3 + ): + try: + left, bottom, right, top = page_object.get_pos() + image_area += max(0.0, right - left) * max(0.0, top - bottom) + finally: + _close_pdfium_child(page_object) - coverage_ratio = min(image_area / page_area, 1.0) if page_area > 0 else 0.0 - if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD: - high_image_coverage_pages += 1 + coverage_ratio = ( + min(image_area / page_area, 1.0) if page_area > 0 else 0.0 + ) + if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD: + high_image_coverage_pages += 1 + finally: + _close_pdfium_child(page) if not page_indices: return 0.0 From 949b1b9223bf193cb6d1cb47f60c49b68c527508 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 17:01:49 +0800 Subject: [PATCH 18/21] fix: ensure proper closure of PDFium child objects to prevent resource leaks --- .../hybrid_model_output_to_middle_json.py | 28 ++-- .../pipeline/model_json_to_middle_json.py | 32 ++-- mineru/backend/pipeline/pipeline_analyze.py | 158 +++++++++--------- .../vlm/model_output_to_middle_json.py | 12 +- mineru/utils/pdf_classify.py | 20 +-- mineru/utils/pdf_image_tools.py | 45 ++++- mineru/utils/pdf_reader.py | 15 +- mineru/utils/pdf_text_tool.py | 39 +++-- mineru/utils/pdfium_guard.py | 32 +++- mineru/utils/span_pre_proc.py | 151 +++++++++-------- 10 files changed, 304 insertions(+), 228 deletions(-) diff --git a/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py b/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py index c8891561..aa03e4cb 100644 --- a/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py +++ b/mineru/backend/hybrid/hybrid_model_output_to_middle_json.py @@ -21,7 +21,7 @@ from mineru.utils.enum_class import ContentType, BlockType from mineru.utils.hash_utils import bytes_md5 from mineru.utils.ocr_utils import OcrConfidence, rotate_vertical_crop_if_needed from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info -from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard +from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard from mineru.version import __version__ @@ -198,17 +198,21 @@ def append_page_results_to_middle_json( zip(model_list, images_list) ): page_index = page_start_index + offset - with pdfium_guard(): - page = pdf_doc[page_index] - page_info = blocks_to_page_info( - page_model_list, - image_dict, - page, - image_writer, - page_index, - _ocr_enable, - _vlm_ocr_enable, - ) + page = None + try: + with pdfium_guard(): + page = pdf_doc[page_index] + page_info = blocks_to_page_info( + page_model_list, + image_dict, + page, + image_writer, + page_index, + _ocr_enable, + _vlm_ocr_enable, + ) + finally: + close_pdfium_child(page) middle_json["pdf_info"].append(page_info) if progress_bar is not None: progress_bar.update(1) diff --git a/mineru/backend/pipeline/model_json_to_middle_json.py b/mineru/backend/pipeline/model_json_to_middle_json.py index d9700724..7ac043ff 100644 --- a/mineru/backend/pipeline/model_json_to_middle_json.py +++ b/mineru/backend/pipeline/model_json_to_middle_json.py @@ -18,7 +18,7 @@ from mineru.backend.pipeline.pipeline_magic_model import MagicModel from mineru.utils.ocr_utils import OcrConfidence, rotate_vertical_crop_if_needed from mineru.version import __version__ from mineru.utils.hash_utils import bytes_md5 -from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard +from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard def page_model_info_to_page_info(page_model_info, image_dict, page, image_writer, page_index, ocr_enable=False): @@ -77,20 +77,24 @@ def append_page_model_infos_to_middle_json( ): for offset, (page_model_info, image_dict) in enumerate(zip(page_model_infos, images_list)): page_index = page_start_index + offset - with pdfium_guard(): - page = pdf_doc[page_index] - page_info = page_model_info_to_page_info( - copy.deepcopy(page_model_info), - image_dict, - page, - image_writer, - page_index, - ocr_enable=ocr_enable, - ) - if page_info is None: + page = None + try: with pdfium_guard(): - page_w, page_h = map(int, pdf_doc[page_index].get_size()) - page_info = make_page_info_dict([], page_index, page_w, page_h, []) + page = pdf_doc[page_index] + page_info = page_model_info_to_page_info( + copy.deepcopy(page_model_info), + image_dict, + page, + image_writer, + page_index, + ocr_enable=ocr_enable, + ) + if page_info is None: + with pdfium_guard(): + page_w, page_h = map(int, page.get_size()) + page_info = make_page_info_dict([], page_index, page_w, page_h, []) + finally: + close_pdfium_child(page) middle_json["pdf_info"].append(page_info) if progress_bar is not None: progress_bar.update(1) diff --git a/mineru/backend/pipeline/pipeline_analyze.py b/mineru/backend/pipeline/pipeline_analyze.py index 6d31b798..1fcb517a 100644 --- a/mineru/backend/pipeline/pipeline_analyze.py +++ b/mineru/backend/pipeline/pipeline_analyze.py @@ -168,53 +168,56 @@ def doc_analyze_streaming( raise ValueError("pdf_bytes_list, image_writer_list, and lang_list must have the same length") doc_contexts = [] - total_pages = 0 - for doc_index, (pdf_bytes, image_writer, lang) in enumerate( - zip(pdf_bytes_list, image_writer_list, lang_list) - ): - _ocr_enable = _get_ocr_enable(pdf_bytes, parse_method) - pdf_doc = open_pdfium_document(pdfium.PdfDocument, pdf_bytes) - page_count = get_pdfium_document_page_count(pdf_doc) - total_pages += page_count - doc_contexts.append( - { - 'doc_index': doc_index, - 'pdf_bytes': pdf_bytes, - 'pdf_doc': pdf_doc, - 'page_count': page_count, - 'next_page_idx': 0, - 'middle_json': init_middle_json(), - 'model_list': [], - 'image_writer': image_writer, - 'lang': lang, - 'ocr_enable': _ocr_enable, - 'closed': False, - } + try: + total_pages = 0 + for doc_index, (pdf_bytes, image_writer, lang) in enumerate( + zip(pdf_bytes_list, image_writer_list, lang_list) + ): + _ocr_enable = _get_ocr_enable(pdf_bytes, parse_method) + pdf_doc = open_pdfium_document(pdfium.PdfDocument, pdf_bytes) + try: + page_count = get_pdfium_document_page_count(pdf_doc) + context = { + 'doc_index': doc_index, + 'pdf_bytes': pdf_bytes, + 'pdf_doc': pdf_doc, + 'page_count': page_count, + 'next_page_idx': 0, + 'middle_json': init_middle_json(), + 'model_list': [], + 'image_writer': image_writer, + 'lang': lang, + 'ocr_enable': _ocr_enable, + 'closed': False, + } + except Exception: + close_pdfium_document(pdf_doc) + raise + total_pages += page_count + doc_contexts.append(context) + + if total_pages == 0: + _emit_zero_page_contexts( + doc_contexts, + on_doc_ready, + client_side_output_generation=client_side_output_generation, + ) + return + + window_size = get_processing_window_size(default=64) + total_batches = (total_pages + window_size - 1) // window_size + logger.info( + f'Pipeline processing-window multi-file run. doc_count={len(doc_contexts)}, ' + f'total_pages={total_pages}, window_size={window_size}, total_batches={total_batches}' ) - if total_pages == 0: _emit_zero_page_contexts( doc_contexts, on_doc_ready, client_side_output_generation=client_side_output_generation, ) - return - - window_size = get_processing_window_size(default=64) - total_batches = (total_pages + window_size - 1) // window_size - logger.info( - f'Pipeline processing-window multi-file run. doc_count={len(doc_contexts)}, ' - f'total_pages={total_pages}, window_size={window_size}, total_batches={total_batches}' - ) - - _emit_zero_page_contexts( - doc_contexts, - on_doc_ready, - client_side_output_generation=client_side_output_generation, - ) - processed_pages = 0 - infer_start = time.time() - try: + processed_pages = 0 + infer_start = time.time() progress_bar = None last_append_end_time = None try: @@ -264,45 +267,48 @@ def doc_analyze_streaming( f'batch_pages={len(batch_images)}, doc_slices={_format_doc_slices(batch_slices)}' ) - batch_results = batch_image_analyze( - batch_images, - formula_enable=formula_enable, - table_enable=table_enable, - ) - if progress_bar is None: - progress_bar = tqdm(total=total_pages, desc="Processing pages") - else: - exclude_progress_bar_idle_time( - progress_bar, - last_append_end_time, - now=time.time(), + try: + batch_results = batch_image_analyze( + batch_images, + formula_enable=formula_enable, + table_enable=table_enable, ) - - result_offset = 0 - for context, images_list, page_start, take_count in batch_payloads: - result_slice = batch_results[result_offset: result_offset + take_count] - append_batch_results_to_middle_json( - context['middle_json'], - result_slice, - images_list, - context['pdf_doc'], - context['image_writer'], - page_start_index=page_start, - ocr_enable=context['ocr_enable'], - model_list=context['model_list'], - progress_bar=progress_bar, - ) - result_offset += take_count - _close_images(images_list) - images_list.clear() - - if context['next_page_idx'] >= context['page_count'] and not context['closed']: - _finalize_processing_window_context( - context, - on_doc_ready, - client_side_output_generation=client_side_output_generation, + if progress_bar is None: + progress_bar = tqdm(total=total_pages, desc="Processing pages") + else: + exclude_progress_bar_idle_time( + progress_bar, + last_append_end_time, + now=time.time(), ) + result_offset = 0 + for context, images_list, page_start, take_count in batch_payloads: + result_slice = batch_results[result_offset: result_offset + take_count] + append_batch_results_to_middle_json( + context['middle_json'], + result_slice, + images_list, + context['pdf_doc'], + context['image_writer'], + page_start_index=page_start, + ocr_enable=context['ocr_enable'], + model_list=context['model_list'], + progress_bar=progress_bar, + ) + result_offset += take_count + + if context['next_page_idx'] >= context['page_count'] and not context['closed']: + _finalize_processing_window_context( + context, + on_doc_ready, + client_side_output_generation=client_side_output_generation, + ) + finally: + for _context, images_list, _page_start, _take_count in batch_payloads: + _close_images(images_list) + images_list.clear() + last_append_end_time = time.time() processed_pages += len(batch_images) finally: diff --git a/mineru/backend/vlm/model_output_to_middle_json.py b/mineru/backend/vlm/model_output_to_middle_json.py index e0073505..29c87727 100644 --- a/mineru/backend/vlm/model_output_to_middle_json.py +++ b/mineru/backend/vlm/model_output_to_middle_json.py @@ -16,7 +16,7 @@ from mineru.utils.cut_image import cut_image_and_table from mineru.utils.enum_class import ContentType from mineru.utils.hash_utils import bytes_md5 from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info -from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard +from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard from mineru.version import __version__ @@ -91,9 +91,13 @@ def append_page_blocks_to_middle_json( ): for offset, (page_blocks, image_dict) in enumerate(zip(model_output_blocks_list, images_list)): page_index = page_start_index + offset - with pdfium_guard(): - page = pdf_doc[page_index] - page_info = blocks_to_page_info(page_blocks, image_dict, page, image_writer, page_index) + page = None + try: + with pdfium_guard(): + page = pdf_doc[page_index] + page_info = blocks_to_page_info(page_blocks, image_dict, page, image_writer, page_index) + finally: + close_pdfium_child(page) middle_json["pdf_info"].append(page_info) if progress_bar is not None: progress_bar.update(1) diff --git a/mineru/utils/pdf_classify.py b/mineru/utils/pdf_classify.py index e860eecb..9abd758a 100644 --- a/mineru/utils/pdf_classify.py +++ b/mineru/utils/pdf_classify.py @@ -8,6 +8,7 @@ import pypdfium2.raw as pdfium_c from loguru import logger from pypdf import PdfReader from mineru.utils.pdfium_guard import ( + close_pdfium_child, close_pdfium_document, open_pdfium_document, pdfium_guard, @@ -225,20 +226,11 @@ def get_extreme_aspect_ratio_page_pdfium( if aspect_ratio > max_page_aspect_ratio: return page_index, aspect_ratio finally: - _close_pdfium_child(page) + close_pdfium_child(page) return None, None -def _close_pdfium_child(pdfium_obj) -> None: - """显式关闭 PDFium 子对象,避免依赖 weakref/finalizer 延迟释放 native 资源。""" - if pdfium_obj is None: - return - close = getattr(pdfium_obj, "close", None) - if callable(close): - close() - - def _collect_pdfium_text_sample_from_page(page_index, page): """从单页 PDFium 对象提取纯 Python 文本统计,并在调用方释放子对象。""" text_page = None @@ -286,7 +278,7 @@ def _collect_pdfium_text_sample_from_page(page_index, page): "font_name_counts": font_name_counts, } finally: - _close_pdfium_child(text_page) + close_pdfium_child(text_page) def _collect_pdfium_text_samples(pdf_doc, page_indices): @@ -302,7 +294,7 @@ def _collect_pdfium_text_samples(pdf_doc, page_indices): _collect_pdfium_text_sample_from_page(page_index, page) ) finally: - _close_pdfium_child(page) + close_pdfium_child(page) return text_samples @@ -669,7 +661,7 @@ def get_high_image_coverage_ratio_pdfium(pdf_doc, page_indices): left, bottom, right, top = page_object.get_pos() image_area += max(0.0, right - left) * max(0.0, top - bottom) finally: - _close_pdfium_child(page_object) + close_pdfium_child(page_object) coverage_ratio = ( min(image_area / page_area, 1.0) if page_area > 0 else 0.0 @@ -677,7 +669,7 @@ def get_high_image_coverage_ratio_pdfium(pdf_doc, page_indices): if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD: high_image_coverage_pages += 1 finally: - _close_pdfium_child(page) + close_pdfium_child(page) if not page_indices: return 0.0 diff --git a/mineru/utils/pdf_image_tools.py b/mineru/utils/pdf_image_tools.py index d06c7a05..de013d9c 100644 --- a/mineru/utils/pdf_image_tools.py +++ b/mineru/utils/pdf_image_tools.py @@ -21,6 +21,7 @@ from mineru.utils.enum_class import ImageType from mineru.utils.hash_utils import str_sha256 from mineru.utils.pdf_page_id import get_end_page_id from mineru.utils.pdfium_guard import ( + close_pdfium_child, close_pdfium_document, get_pdfium_document_page_count, open_pdfium_document, @@ -66,7 +67,10 @@ def pdf_page_to_image( "scale": scale, } if image_type == ImageType.BASE64: - image_dict["img_base64"] = image_to_b64str(pil_img) + try: + image_dict["img_base64"] = image_to_b64str(pil_img) + finally: + pil_img.close() else: image_dict["img_pil"] = pil_img @@ -82,6 +86,18 @@ def _load_images_from_pdf_worker( ) +def _close_image_dicts(images_list) -> None: + """关闭 image dict 中的 PIL 图片,供异常清理路径释放已生成的图像资源。""" + for image_dict in images_list or []: + pil_img = image_dict.get("img_pil") + if pil_img is None: + continue + try: + pil_img.close() + except Exception: + pass + + def _calculate_render_process_count(total_pages: int, threads: int, cpu_count=None) -> int: requested_threads = max(1, threads) available_cpus = max(1, cpu_count if cpu_count is not None else (os.cpu_count() or 1)) @@ -277,6 +293,7 @@ def _load_images_from_pdf_bytes_range( executor = _get_pdf_render_executor() recycle_executor = False + collected_image_lists = [] try: futures = [] future_to_range = {} @@ -305,6 +322,7 @@ def _load_images_from_pdf_bytes_range( for future in futures: range_start = future_to_range[future] images_list = future.result() + collected_image_lists.append(images_list) all_results.append((range_start, images_list)) all_results.sort(key=lambda x: x[0]) @@ -312,10 +330,15 @@ def _load_images_from_pdf_bytes_range( for _, imgs in all_results: images_list.extend(imgs) + collected_image_lists.clear() return images_list except BrokenProcessPool: recycle_executor = True raise + except Exception: + for images_list in collected_image_lists: + _close_image_dicts(images_list) + raise finally: if recycle_executor: logger.warning("Recycling persistent PDF render executor after render failure") @@ -410,9 +433,13 @@ def load_images_from_pdf_core( for index in range(start_page_id, end_page_id + 1): # logger.debug(f"Converting page {index}/{pdf_page_num} to image") - page = pdf_doc[index] - image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) - images_list.append(image_dict) + page = None + try: + page = pdf_doc[index] + image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) + images_list.append(image_dict) + finally: + close_pdfium_child(page) finally: close_pdfium_document(pdf_doc) @@ -446,9 +473,13 @@ def load_images_from_pdf_doc( images_list = [] with pdfium_guard(): for index in range(start_page_id, normalized_end_page_id + 1): - page = pdf_doc[index] - image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) - images_list.append(image_dict) + page = None + try: + page = pdf_doc[index] + image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type) + images_list.append(image_dict) + finally: + close_pdfium_child(page) return images_list diff --git a/mineru/utils/pdf_reader.py b/mineru/utils/pdf_reader.py index f34b382e..c6a715e6 100644 --- a/mineru/utils/pdf_reader.py +++ b/mineru/utils/pdf_reader.py @@ -20,13 +20,16 @@ def page_to_image( if (long_side_length*scale) > max_width_or_height: scale = max_width_or_height / long_side_length - bitmap: PdfBitmap = page.render(scale=scale) # type: ignore - - image = bitmap.to_pil() + bitmap: PdfBitmap | None = None try: - bitmap.close() - except Exception as e: - logger.error(f"Failed to close bitmap: {e}") + bitmap = page.render(scale=scale) # type: ignore + image = bitmap.to_pil().copy() + finally: + if bitmap is not None: + try: + bitmap.close() + except Exception as e: + logger.error(f"Failed to close bitmap: {e}") return image, scale diff --git a/mineru/utils/pdf_text_tool.py b/mineru/utils/pdf_text_tool.py index a1321de1..7cb495fd 100644 --- a/mineru/utils/pdf_text_tool.py +++ b/mineru/utils/pdf_text_tool.py @@ -6,7 +6,7 @@ import pypdfium2 as pdfium from pdftext.pdf.chars import deduplicate_chars, get_chars from pdftext.pdf.pages import assign_scripts, get_blocks, get_lines, get_spans -from mineru.utils.pdfium_guard import pdfium_guard +from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard def get_page( @@ -39,25 +39,30 @@ def get_page_chars( page_char_count: int | None = None, ) -> dict: """轻量读取页面字符坐标,供只需要 char 级信息的路径复用。""" - with pdfium_guard(): - if textpage is None: - textpage = page.get_textpage() - page_bbox: List[float] = page.get_bbox() - page_width = math.ceil(abs(page_bbox[2] - page_bbox[0])) - page_height = math.ceil(abs(page_bbox[1] - page_bbox[3])) + owns_textpage = textpage is None + try: + with pdfium_guard(): + if textpage is None: + textpage = page.get_textpage() + page_bbox: List[float] = page.get_bbox() + page_width = math.ceil(abs(page_bbox[2] - page_bbox[0])) + page_height = math.ceil(abs(page_bbox[1] - page_bbox[3])) - page_rotation = 0 - try: - page_rotation = page.get_rotation() - except Exception: - pass + page_rotation = 0 + try: + page_rotation = page.get_rotation() + except Exception: + pass - if page_char_count is None: - page_char_count = textpage.count_chars() + if page_char_count is None: + page_char_count = textpage.count_chars() - chars = deduplicate_chars( - get_chars(textpage, page_bbox, page_rotation, quote_loosebox) - ) + chars = deduplicate_chars( + get_chars(textpage, page_bbox, page_rotation, quote_loosebox) + ) + finally: + if owns_textpage: + close_pdfium_child(textpage) return { "bbox": page_bbox, diff --git a/mineru/utils/pdfium_guard.py b/mineru/utils/pdfium_guard.py index 08e90fc6..87dbb281 100644 --- a/mineru/utils/pdfium_guard.py +++ b/mineru/utils/pdfium_guard.py @@ -4,6 +4,8 @@ from io import BytesIO from contextlib import contextmanager from typing import Any, Callable, Sequence, TypeVar +from loguru import logger + from mineru.utils.pdf_page_id import get_end_page_id @@ -39,6 +41,27 @@ def close_pdfium_document(pdf_doc) -> None: pdf_doc.close() +def close_pdfium_child(pdfium_obj) -> None: + """显式关闭 PDFium 子对象,避免依赖 weakref/finalizer 延迟释放 native 资源。""" + if pdfium_obj is None: + return + close = getattr(pdfium_obj, "close", None) + if callable(close): + with pdfium_guard(): + close() + + +def close_pdfium_objects_safely(*pdfium_objs, owner: str = "pdfium cleanup") -> None: + """清理多个 PDFium 对象时逐个尝试关闭,避免前一个关闭失败阻断后续对象释放。""" + for pdfium_obj in pdfium_objs: + if pdfium_obj is None: + continue + try: + close_pdfium_child(pdfium_obj) + except Exception as exc: + logger.warning(f"Failed to close PDFium object during {owner}: {exc}") + + def rewrite_pdf_bytes_with_pdfium( src_pdf_bytes: bytes, start_page_id: int = 0, @@ -79,7 +102,8 @@ def rewrite_pdf_bytes_with_pdfium( output_doc.save(output_buffer) return output_buffer.getvalue() finally: - if output_doc is not None: - close_pdfium_document(output_doc) - if pdf_doc is not None: - close_pdfium_document(pdf_doc) + close_pdfium_objects_safely( + output_doc, + pdf_doc, + owner="rewrite_pdf_bytes_with_pdfium", + ) diff --git a/mineru/utils/span_pre_proc.py b/mineru/utils/span_pre_proc.py index 4a57eed5..7648d815 100644 --- a/mineru/utils/span_pre_proc.py +++ b/mineru/utils/span_pre_proc.py @@ -12,7 +12,7 @@ from mineru.utils.boxbase import calculate_overlap_area_in_bbox1_area_ratio from mineru.utils.enum_class import BlockType, ContentType from mineru.utils.pdf_image_tools import get_crop_img from mineru.utils.pdf_text_tool import get_lines_from_chars, get_page_chars -from mineru.utils.pdfium_guard import pdfium_guard +from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard MAX_NATIVE_TEXT_CHARS_PER_PAGE = 65535 @@ -35,91 +35,94 @@ def txt_spans_extract(pdf_page, spans, pil_img, scale, all_bboxes, all_discarded page_char_count = None textpage = None try: - with pdfium_guard(): - textpage = pdf_page.get_textpage() - page_char_count = textpage.count_chars() - except Exception as exc: - logger.debug(f"Failed to get page char count before txt extraction: {exc}") + try: + with pdfium_guard(): + textpage = pdf_page.get_textpage() + page_char_count = textpage.count_chars() + except Exception as exc: + logger.debug(f"Failed to get page char count before txt extraction: {exc}") - if page_char_count is not None and page_char_count > MAX_NATIVE_TEXT_CHARS_PER_PAGE: - logger.info( - "Fallback to post-OCR in txt_spans_extract due to high char count: " - f"count_chars={page_char_count}" + if page_char_count is not None and page_char_count > MAX_NATIVE_TEXT_CHARS_PER_PAGE: + logger.info( + "Fallback to post-OCR in txt_spans_extract due to high char count: " + f"count_chars={page_char_count}" + ) + need_ocr_spans = [ + span for span in spans if span.get('type') == ContentType.TEXT + ] + return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale) + + page_chars = get_page_chars( + pdf_page, + textpage=textpage, + page_char_count=page_char_count, ) - need_ocr_spans = [ - span for span in spans if span.get('type') == ContentType.TEXT + page_all_chars = [ + char for char in page_chars['chars'] + if _is_supported_rotation(char['rotation']) ] - return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale) - page_chars = get_page_chars( - pdf_page, - textpage=textpage, - page_char_count=page_char_count, - ) - page_all_chars = [ - char for char in page_chars['chars'] - if _is_supported_rotation(char['rotation']) - ] + # 计算所有sapn的高度的中位数 + span_height_list = [] + for span in spans: + if span['type'] in [ContentType.TEXT]: + span_height = span['bbox'][3] - span['bbox'][1] + span['height'] = span_height + span['width'] = span['bbox'][2] - span['bbox'][0] + span_height_list.append(span_height) + if len(span_height_list) == 0: + return spans + else: + median_span_height = statistics.median(span_height_list) - # 计算所有sapn的高度的中位数 - span_height_list = [] - for span in spans: - if span['type'] in [ContentType.TEXT]: - span_height = span['bbox'][3] - span['bbox'][1] - span['height'] = span_height - span['width'] = span['bbox'][2] - span['bbox'][0] - span_height_list.append(span_height) - if len(span_height_list) == 0: - return spans - else: - median_span_height = statistics.median(span_height_list) + useful_spans = [] + unuseful_spans = [] + # 纵向span的两个特征:1. 高度超过多个line 2. 高宽比超过某个值 + vertical_spans = [] + for span in spans: + if span['type'] in [ContentType.TEXT]: + for block in all_bboxes + all_discarded_blocks: + if block[7] in [BlockType.IMAGE_BODY, BlockType.TABLE_BODY, BlockType.INTERLINE_EQUATION]: + continue + if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5: + if span['height'] > median_span_height * 2.3 and span['height'] > span['width'] * 2.3: + vertical_spans.append(span) + elif block in all_bboxes: + useful_spans.append(span) + else: + unuseful_spans.append(span) + break - useful_spans = [] - unuseful_spans = [] - # 纵向span的两个特征:1. 高度超过多个line 2. 高宽比超过某个值 - vertical_spans = [] - for span in spans: - if span['type'] in [ContentType.TEXT]: - for block in all_bboxes + all_discarded_blocks: - if block[7] in [BlockType.IMAGE_BODY, BlockType.TABLE_BODY, BlockType.INTERLINE_EQUATION]: - continue - if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5: - if span['height'] > median_span_height * 2.3 and span['height'] > span['width'] * 2.3: - vertical_spans.append(span) - elif block in all_bboxes: - useful_spans.append(span) - else: - unuseful_spans.append(span) - break + """垂直的span框直接用line进行填充""" + if len(vertical_spans) > 0: + page_all_lines = [ + line for line in get_lines_from_chars(page_chars['chars']) + if _is_supported_rotation(line['rotation']) + ] + for pdfium_line in page_all_lines: + for span in vertical_spans: + if calculate_overlap_area_in_bbox1_area_ratio(pdfium_line['bbox'].bbox, span['bbox']) > 0.5: + for pdfium_span in pdfium_line['spans']: + span['content'] += pdfium_span['text'] + break - """垂直的span框直接用line进行填充""" - if len(vertical_spans) > 0: - page_all_lines = [ - line for line in get_lines_from_chars(page_chars['chars']) - if _is_supported_rotation(line['rotation']) - ] - for pdfium_line in page_all_lines: for span in vertical_spans: - if calculate_overlap_area_in_bbox1_area_ratio(pdfium_line['bbox'].bbox, span['bbox']) > 0.5: - for pdfium_span in pdfium_line['spans']: - span['content'] += pdfium_span['text'] - break + if len(span['content']) == 0: + spans.remove(span) - for span in vertical_spans: - if len(span['content']) == 0: - spans.remove(span) + """水平的span框先用char填充,再用ocr填充空的span框""" + new_spans = [] - """水平的span框先用char填充,再用ocr填充空的span框""" - new_spans = [] + for span in useful_spans + unuseful_spans: + if span['type'] in [ContentType.TEXT]: + span['chars'] = [] + new_spans.append(span) - for span in useful_spans + unuseful_spans: - if span['type'] in [ContentType.TEXT]: - span['chars'] = [] - new_spans.append(span) + need_ocr_spans = fill_char_in_spans(new_spans, page_all_chars, median_span_height) - need_ocr_spans = fill_char_in_spans(new_spans, page_all_chars, median_span_height) - - return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale) + return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale) + finally: + close_pdfium_child(textpage) def _is_supported_rotation(rotation) -> bool: From 86b84b9f27e68e79c7435fb5ecc164373724e942 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 17:30:01 +0800 Subject: [PATCH 19/21] fix: ensure proper closure of PDFium child objects to prevent resource leaks --- mineru/backend/pipeline/model_init.py | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/mineru/backend/pipeline/model_init.py b/mineru/backend/pipeline/model_init.py index f696dc00..d4a5b504 100644 --- a/mineru/backend/pipeline/model_init.py +++ b/mineru/backend/pipeline/model_init.py @@ -25,6 +25,10 @@ PIPELINE_LAYOUT_INFERENCE_LOCK = threading.RLock() PIPELINE_MFR_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_DET_INFERENCE_LOCK = threading.RLock() PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock() +# 临时关闭 pipeline/hybrid 共享推理阶段锁;需要回滚实验时可通过环境变量重新打开。 +PIPELINE_INFERENCE_LOCKS_ENABLED = os.getenv( + 'MINERU_ENABLE_PIPELINE_INFERENCE_LOCKS', 'False' +).lower() in ['true', '1', 'yes'] # 推理准入门闩:清理线程等待静默区时,阻止新的推理调用继续插队进入阶段锁。 PIPELINE_INFERENCE_ADMISSION_LOCK = threading.RLock() PIPELINE_INFERENCE_STAGE_LOCKS = ( @@ -36,7 +40,10 @@ PIPELINE_INFERENCE_STAGE_LOCKS = ( def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs): - """在指定推理锁内执行真实 native 模型调用。""" + """按实验开关决定是否在指定推理锁内执行真实 native 模型调用。""" + if not PIPELINE_INFERENCE_LOCKS_ENABLED: + return inference_callable(*args, **kwargs) + with PIPELINE_INFERENCE_ADMISSION_LOCK: inference_lock.acquire() try: @@ -47,7 +54,11 @@ def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs @contextmanager def pipeline_inference_quiet_zone(): - """进入 pipeline/hybrid 共享模型推理静默区,等待四类阶段推理全部退出。""" + """进入 pipeline/hybrid 共享模型推理静默区;锁关闭时保持直通。""" + if not PIPELINE_INFERENCE_LOCKS_ENABLED: + yield + return + acquired_locks = [] with PIPELINE_INFERENCE_ADMISSION_LOCK: try: From 33ebaa38685a2be29567a4511a95a58c02619183 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 17:51:32 +0800 Subject: [PATCH 20/21] refactor: simplify inference locking mechanism and update docstrings --- mineru/backend/pipeline/model_init.py | 42 ++++----------------------- 1 file changed, 5 insertions(+), 37 deletions(-) diff --git a/mineru/backend/pipeline/model_init.py b/mineru/backend/pipeline/model_init.py index d4a5b504..12738f46 100644 --- a/mineru/backend/pipeline/model_init.py +++ b/mineru/backend/pipeline/model_init.py @@ -1,7 +1,6 @@ # Copyright (c) Opendatalab. All rights reserved. import os import threading -from contextlib import contextmanager import torch from loguru import logger @@ -29,14 +28,6 @@ PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock() PIPELINE_INFERENCE_LOCKS_ENABLED = os.getenv( 'MINERU_ENABLE_PIPELINE_INFERENCE_LOCKS', 'False' ).lower() in ['true', '1', 'yes'] -# 推理准入门闩:清理线程等待静默区时,阻止新的推理调用继续插队进入阶段锁。 -PIPELINE_INFERENCE_ADMISSION_LOCK = threading.RLock() -PIPELINE_INFERENCE_STAGE_LOCKS = ( - PIPELINE_LAYOUT_INFERENCE_LOCK, - PIPELINE_MFR_INFERENCE_LOCK, - PIPELINE_OCR_DET_INFERENCE_LOCK, - PIPELINE_OCR_REC_INFERENCE_LOCK, -) def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs): @@ -44,56 +35,33 @@ def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs if not PIPELINE_INFERENCE_LOCKS_ENABLED: return inference_callable(*args, **kwargs) - with PIPELINE_INFERENCE_ADMISSION_LOCK: - inference_lock.acquire() - try: + with inference_lock: return inference_callable(*args, **kwargs) - finally: - inference_lock.release() - - -@contextmanager -def pipeline_inference_quiet_zone(): - """进入 pipeline/hybrid 共享模型推理静默区;锁关闭时保持直通。""" - if not PIPELINE_INFERENCE_LOCKS_ENABLED: - yield - return - - acquired_locks = [] - with PIPELINE_INFERENCE_ADMISSION_LOCK: - try: - for inference_lock in PIPELINE_INFERENCE_STAGE_LOCKS: - inference_lock.acquire() - acquired_locks.append(inference_lock) - yield - finally: - for inference_lock in reversed(acquired_locks): - inference_lock.release() def run_layout_inference(inference_callable, *args, **kwargs): - """在共享 Layout 推理锁内执行模型调用。""" + """按实验开关执行共享 Layout 模型调用。""" return _run_with_inference_lock( PIPELINE_LAYOUT_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_mfr_inference(inference_callable, *args, **kwargs): - """在共享 MFR 推理锁内执行模型调用。""" + """按实验开关执行共享 MFR 模型调用。""" return _run_with_inference_lock( PIPELINE_MFR_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_ocr_det_inference(inference_callable, *args, **kwargs): - """在共享 OCR det 推理锁内执行模型调用。""" + """按实验开关执行共享 OCR det 模型调用。""" return _run_with_inference_lock( PIPELINE_OCR_DET_INFERENCE_LOCK, inference_callable, *args, **kwargs ) def run_ocr_rec_inference(inference_callable, *args, **kwargs): - """在共享 OCR rec 推理锁内执行模型调用。""" + """按实验开关执行共享 OCR rec 模型调用。""" return _run_with_inference_lock( PIPELINE_OCR_REC_INFERENCE_LOCK, inference_callable, *args, **kwargs ) From f79b4356c3b70dd0c9ba0a4b18ad5c1a2b50483c Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 2 Jun 2026 19:34:32 +0800 Subject: [PATCH 21/21] feat: add functionality to skip broken PDF pages during rewrite process --- mineru/cli/common.py | 45 ++++++++++++++++++++++++++++++++++-- mineru/utils/pdfium_guard.py | 43 ++++++++++++++++++++++++++++++++++ 2 files changed, 86 insertions(+), 2 deletions(-) diff --git a/mineru/cli/common.py b/mineru/cli/common.py index 3d69de8e..f029dd76 100644 --- a/mineru/cli/common.py +++ b/mineru/cli/common.py @@ -23,7 +23,10 @@ from mineru.backend.vlm.vlm_analyze import aio_doc_analyze as aio_vlm_doc_analyz from mineru.backend.office.pptx_analyze import office_pptx_analyze from mineru.backend.office.xlsx_analyze import office_xlsx_analyze from mineru.backend.office.docx_analyze import office_docx_analyze -from mineru.utils.pdfium_guard import rewrite_pdf_bytes_with_pdfium +from mineru.utils.pdfium_guard import ( + get_loadable_pdfium_page_indices, + rewrite_pdf_bytes_with_pdfium, +) os.environ["TORCH_CUDNN_V8_API_DISABLED"] = "1" if os.getenv("MINERU_LMDEPLOY_DEVICE", "") == "maca": @@ -190,12 +193,50 @@ def convert_pdf_bytes_to_bytes(pdf_bytes, start_page_id=0, end_page_id=None): ) if rebuilt_pdf_bytes: return rebuilt_pdf_bytes - logger.warning("PDFium rewrite returned empty bytes, using original PDF bytes.") + logger.warning( + "PDFium rewrite returned empty bytes, trying to skip broken pages." + ) except Exception as fallback_error: logger.warning( f"Error in converting PDF bytes with pdfium: {fallback_error}, " + "trying to skip broken pages." + ) + + try: + loadable_page_indices, broken_page_indices = get_loadable_pdfium_page_indices( + pdf_bytes, + start_page_id=start_page_id, + end_page_id=end_page_id, + ) + if broken_page_indices: + skipped_pages = [page_index + 1 for page_index in broken_page_indices] + logger.warning( + f"Skipped broken PDF pages during PDFium rewrite: {skipped_pages}" + ) + if not loadable_page_indices: + logger.warning( + "PDFium skip-broken-page rewrite found no loadable pages, " + "using original PDF bytes." + ) + return pdf_bytes + + rebuilt_pdf_bytes = rewrite_pdf_bytes_with_pdfium( + pdf_bytes, + start_page_id=start_page_id, + end_page_id=end_page_id, + page_indices=loadable_page_indices, + ) + if rebuilt_pdf_bytes: + return rebuilt_pdf_bytes + logger.warning( + "PDFium skip-broken-page rewrite returned empty bytes, " "using original PDF bytes." ) + except Exception as fallback_error: + logger.warning( + "Error in converting PDF bytes with skip-broken-page fallback: " + f"{fallback_error}, using original PDF bytes." + ) return pdf_bytes diff --git a/mineru/utils/pdfium_guard.py b/mineru/utils/pdfium_guard.py index 87dbb281..97918f29 100644 --- a/mineru/utils/pdfium_guard.py +++ b/mineru/utils/pdfium_guard.py @@ -62,6 +62,49 @@ def close_pdfium_objects_safely(*pdfium_objs, owner: str = "pdfium cleanup") -> logger.warning(f"Failed to close PDFium object during {owner}: {exc}") +def get_loadable_pdfium_page_indices( + src_pdf_bytes: bytes, + start_page_id: int = 0, + end_page_id: int | None = None, +) -> tuple[list[int], list[int]]: + """逐页探测 PDFium 可加载页面,返回可保留页和损坏页的 0-based 索引。""" + import pypdfium2 as pdfium + + loadable_page_indices = [] + broken_page_indices = [] + pdf_doc = None + + try: + with pdfium_guard(): + pdf_doc = pdfium.PdfDocument(src_pdf_bytes) + total_page_count = len(pdf_doc) + if total_page_count == 0: + return [], [] + + normalized_start_page_id = max(0, start_page_id) + normalized_end_page_id = get_end_page_id(end_page_id, total_page_count) + if normalized_start_page_id > normalized_end_page_id: + return [], [] + + for page_index in range( + normalized_start_page_id, + normalized_end_page_id + 1, + ): + page = None + try: + page = pdf_doc[page_index] + page.get_size() + loadable_page_indices.append(page_index) + except Exception: + broken_page_indices.append(page_index) + finally: + close_pdfium_child(page) + finally: + close_pdfium_document(pdf_doc) + + return loadable_page_indices, broken_page_indices + + def rewrite_pdf_bytes_with_pdfium( src_pdf_bytes: bytes, start_page_id: int = 0,