Merge pull request #5063 from opendatalab/dev

This commit is contained in:
Xiaomeng Zhao
2026-06-02 19:44:16 +08:00
committed by GitHub
21 changed files with 842 additions and 322 deletions
+7
View File
@@ -21,6 +21,10 @@ Options:
-e, --end INTEGER Ending page number for parsing (0-based)
-f, --formula BOOLEAN Enable formula parsing (default: enabled)
-t, --table BOOLEAN Enable table parsing (default: enabled)
--client-side-output-generation BOOLEAN
Generate Markdown and content lists locally
from server-returned middle JSON, images, and
original files (default: disabled)
--help Show help information
```
> [!TIP]
@@ -59,6 +63,9 @@ Options:
starts a reusable local mineru-api service.
--enable-vlm-preload BOOLEAN Preload the local VLM model when gradio
starts a local mineru-api service.
--client-side-output-generation BOOLEAN
Generate Markdown and content lists locally
from server-returned middle JSON.
--latex-delimiters-type [a|b|all]
Set the type of LaTeX delimiters to use in
Markdown rendering: 'a' for type '$', 'b' for
+6
View File
@@ -21,6 +21,9 @@ Options:
-e, --end INTEGER 结束解析的页码(从 0 开始)
-f, --formula BOOLEAN 是否启用公式解析(默认开启)
-t, --table BOOLEAN 是否启用表格解析(默认开启)
--client-side-output-generation BOOLEAN
在客户端基于服务端返回的 middle JSON、图片与原文件
生成 Markdown 和 content list(默认关闭)
--help 显示帮助信息
```
> [!TIP]
@@ -54,6 +57,9 @@ Options:
mineru-api
--enable-vlm-preload BOOLEAN 在 Gradio 拉起本地 mineru-api 时预加载本地
VLM 模型
--client-side-output-generation BOOLEAN
在客户端基于服务端返回的 middle JSON 生成 Markdown
和 content list
--latex-delimiters-type [a|b|all]
设置在 Markdown 渲染中使用的 LaTeX 分隔符类型
('a' 表示 '$' 类型,'b' 表示 '()[]' 类型,
+27 -7
View File
@@ -19,7 +19,13 @@ from mineru.backend.hybrid.hybrid_model_output_to_middle_json import (
init_middle_json,
)
from mineru.backend.utils.runtime_utils import exclude_progress_bar_idle_time
from mineru.backend.pipeline.model_init import HybridModelSingleton
from mineru.backend.pipeline.model_init import (
HybridModelSingleton,
run_layout_inference,
run_mfr_inference,
run_ocr_det_inference,
run_ocr_rec_inference,
)
from mineru.backend.vlm.vlm_analyze import (
ModelSingleton,
aio_predictor_execution_guard,
@@ -119,8 +125,11 @@ def ocr_det(
page_mfd_res, useful_list
)
bgr_image = cv2.cvtColor(new_image, cv2.COLOR_RGB2BGR)
ocr_res = hybrid_pipeline_model.ocr_model.ocr(
bgr_image, mfd_res=adjusted_mfdetrec_res, rec=False
ocr_res = run_ocr_det_inference(
hybrid_pipeline_model.ocr_model.ocr,
bgr_image,
mfd_res=adjusted_mfdetrec_res,
rec=False,
)[0]
if ocr_res:
ocr_result_list = get_ocr_result_list(
@@ -193,7 +202,11 @@ def ocr_det(
# 批处理检测
det_batch_size = min(len(batch_images), batch_ratio * OCR_DET_BASE_BATCH_SIZE)
batch_results = hybrid_pipeline_model.ocr_model.text_detector.batch_predict(batch_images, det_batch_size)
batch_results = run_ocr_det_inference(
hybrid_pipeline_model.ocr_model.text_detector.batch_predict,
batch_images,
det_batch_size,
)
# 处理批处理结果
for crop_info, (dt_boxes, _) in zip(group_crops, batch_results):
@@ -392,7 +405,8 @@ def _predict_layout_for_title_split(
batch_ratio,
):
"""执行layout小模型检测,专门为Hybrid标题拆分提供页面layout结果。"""
return hybrid_pipeline_model.layout_model.batch_predict(
return run_layout_inference(
hybrid_pipeline_model.layout_model.batch_predict,
images,
batch_size=min(8, batch_ratio * LAYOUT_BASE_BATCH_SIZE),
)
@@ -433,7 +447,8 @@ def _process_ocr_and_formulas(
if inline_formula_enable:
images_mfd_res = _build_inline_formula_inputs(images_layout_res)
# 公式识别
inline_formula_list = hybrid_pipeline_model.mfr_model.batch_predict(
inline_formula_list = run_mfr_inference(
hybrid_pipeline_model.mfr_model.batch_predict,
images_mfd_res,
np_images,
batch_size=batch_ratio * MFR_BASE_BATCH_SIZE,
@@ -473,7 +488,12 @@ def _process_ocr_and_formulas(
img_crop_list.append(ocr_res.pop('np_img'))
if len(img_crop_list) > 0:
# Process OCR
ocr_result_list = hybrid_pipeline_model.ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
ocr_result_list = run_ocr_rec_inference(
hybrid_pipeline_model.ocr_model.ocr,
img_crop_list,
det=False,
tqdm_enable=True,
)[0]
# Verify we have matching counts
assert len(ocr_result_list) == len(need_ocr_list), f'ocr_result_list: {len(ocr_result_list)}, need_ocr_list: {len(need_ocr_list)}'
@@ -14,13 +14,14 @@ from mineru.backend.utils.para_block_utils import (
)
from mineru.backend.hybrid.hybrid_magic_model import MagicModel
from mineru.backend.utils.runtime_utils import cross_page_table_merge
from mineru.backend.pipeline.model_init import run_ocr_rec_inference
from mineru.utils.config_reader import get_table_enable
from mineru.utils.cut_image import cut_image_and_table
from mineru.utils.enum_class import ContentType, BlockType
from mineru.utils.hash_utils import bytes_md5
from mineru.utils.ocr_utils import OcrConfidence, rotate_vertical_crop_if_needed
from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info
from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard
from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard
from mineru.version import __version__
@@ -138,7 +139,12 @@ def _apply_post_ocr(pdf_info_list, hybrid_pipeline_model):
img_crop_list.append(rotate_vertical_crop_if_needed(span['np_img']))
span.pop('np_img')
if len(img_crop_list) > 0:
ocr_res_list = hybrid_pipeline_model.ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
ocr_res_list = run_ocr_rec_inference(
hybrid_pipeline_model.ocr_model.ocr,
img_crop_list,
det=False,
tqdm_enable=True,
)[0]
assert len(ocr_res_list) == len(
need_ocr_list), f'ocr_res_list: {len(ocr_res_list)}, need_ocr_list: {len(need_ocr_list)}'
for index, span in enumerate(need_ocr_list):
@@ -192,17 +198,21 @@ def append_page_results_to_middle_json(
zip(model_list, images_list)
):
page_index = page_start_index + offset
with pdfium_guard():
page = pdf_doc[page_index]
page_info = blocks_to_page_info(
page_model_list,
image_dict,
page,
image_writer,
page_index,
_ocr_enable,
_vlm_ocr_enable,
)
page = None
try:
with pdfium_guard():
page = pdf_doc[page_index]
page_info = blocks_to_page_info(
page_model_list,
image_dict,
page,
image_writer,
page_index,
_ocr_enable,
_vlm_ocr_enable,
)
finally:
close_pdfium_child(page)
middle_json["pdf_info"].append(page_info)
if progress_bar is not None:
progress_bar.update(1)
+36 -11
View File
@@ -9,7 +9,13 @@ from tqdm import tqdm
from collections import defaultdict
import numpy as np
from .model_init import AtomModelSingleton
from .model_init import (
AtomModelSingleton,
run_layout_inference,
run_mfr_inference,
run_ocr_det_inference,
run_ocr_rec_inference,
)
from .model_list import AtomicModel
from ...utils.config_reader import (
get_formula_enable,
@@ -362,9 +368,10 @@ class BatchAnalyze:
np_images = [np.asarray(image) for image, _, _ in images_with_extra_info]
# pp-doclayout_v2
images_layout_res += self.model.layout_model.batch_predict(
images_layout_res += run_layout_inference(
self.model.layout_model.batch_predict,
pil_images,
batch_size=min(8, self.batch_ratio * LAYOUT_BASE_BATCH_SIZE)
batch_size=min(8, self.batch_ratio * LAYOUT_BASE_BATCH_SIZE),
)
# 清理显存
clean_vram(self.model.device, vram_threshold=8)
@@ -380,7 +387,8 @@ class BatchAnalyze:
images_mfd_res.append(page_formula_res)
# 公式识别
images_formula_list = self.model.mfr_model.batch_predict(
images_formula_list = run_mfr_inference(
self.model.mfr_model.batch_predict,
images_mfd_res,
np_images,
batch_size=self.batch_ratio * MFR_BASE_BATCH_SIZE,
@@ -526,7 +534,9 @@ class BatchAnalyze:
if inline_mask_boxes
else bgr_image
)
ocr_result = det_ocr_engine.ocr(det_image, rec=False)[0]
ocr_result = run_ocr_det_inference(
det_ocr_engine.ocr, det_image, rec=False
)[0]
if ocr_result and formula_mask_boxes:
ocr_result = update_det_boxes(ocr_result, formula_mask_boxes)
if ocr_result:
@@ -555,7 +565,13 @@ class BatchAnalyze:
enable_merge_det_boxes=False,
)
cropped_img_list = [item["cropped_img"] for item in rec_img_list]
ocr_res_list = ocr_engine.ocr(cropped_img_list, det=False, tqdm_enable=True, tqdm_desc=f"Table-ocr rec {_lang}")[0]
ocr_res_list = run_ocr_rec_inference(
ocr_engine.ocr,
cropped_img_list,
det=False,
tqdm_enable=True,
tqdm_desc=f"Table-ocr rec {_lang}",
)[0]
# 按照 table_id 将识别结果进行回填
for img_dict, ocr_res in zip(rec_img_list, ocr_res_list):
ocr_text = self._normalize_table_ocr_rec_text(ocr_res[0])
@@ -715,7 +731,9 @@ class BatchAnalyze:
# 批处理检测
det_batch_size = min(len(batch_images), self.batch_ratio * OCR_DET_BASE_BATCH_SIZE)
batch_results = ocr_model.text_detector.batch_predict(batch_images, det_batch_size)
batch_results = run_ocr_det_inference(
ocr_model.text_detector.batch_predict, batch_images, det_batch_size
)
# 处理批处理结果
for crop_info, (dt_boxes, _) in zip(group_crops, batch_results):
@@ -775,8 +793,11 @@ class BatchAnalyze:
bgr_image,
adjusted_mfdetrec_res,
)
ocr_res = ocr_model.ocr(
det_image, mfd_res=adjusted_mfdetrec_res, rec=False
ocr_res = run_ocr_det_inference(
ocr_model.ocr,
det_image,
mfd_res=adjusted_mfdetrec_res,
rec=False,
)[0]
# Integration results
@@ -831,7 +852,9 @@ class BatchAnalyze:
atom_model_name=AtomicModel.OCR,
lang=lang
)
ocr_res_list = ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
ocr_res_list = run_ocr_rec_inference(
ocr_model.ocr, img_crop_list, det=False, tqdm_enable=True
)[0]
# Verify we have matching counts
assert len(ocr_res_list) == len(
@@ -900,7 +923,9 @@ class BatchAnalyze:
)
seal_crop_bgr = cv2.cvtColor(seal_crop_rgb, cv2.COLOR_RGB2BGR)
seal_ocr_res = seal_ocr_model.ocr(seal_crop_bgr, det=True, rec=True)[0]
seal_ocr_res = run_ocr_det_inference(
seal_ocr_model.ocr, seal_crop_bgr, det=True, rec=True
)[0]
if not seal_ocr_res:
continue
+46
View File
@@ -19,6 +19,52 @@ from ...utils.enum_class import ModelPath
from ...utils.models_download_utils import auto_download_and_get_model_root_path
PIPELINE_MODEL_INIT_LOCK = threading.RLock()
# 这些锁保护 pipeline 与 hybrid 共享的 atom model/native 模型推理调用,避免多线程同时进入同一个模型对象。
PIPELINE_LAYOUT_INFERENCE_LOCK = threading.RLock()
PIPELINE_MFR_INFERENCE_LOCK = threading.RLock()
PIPELINE_OCR_DET_INFERENCE_LOCK = threading.RLock()
PIPELINE_OCR_REC_INFERENCE_LOCK = threading.RLock()
# 临时关闭 pipeline/hybrid 共享推理阶段锁;需要回滚实验时可通过环境变量重新打开。
PIPELINE_INFERENCE_LOCKS_ENABLED = os.getenv(
'MINERU_ENABLE_PIPELINE_INFERENCE_LOCKS', 'False'
).lower() in ['true', '1', 'yes']
def _run_with_inference_lock(inference_lock, inference_callable, *args, **kwargs):
"""按实验开关决定是否在指定推理锁内执行真实 native 模型调用。"""
if not PIPELINE_INFERENCE_LOCKS_ENABLED:
return inference_callable(*args, **kwargs)
with inference_lock:
return inference_callable(*args, **kwargs)
def run_layout_inference(inference_callable, *args, **kwargs):
"""按实验开关执行共享 Layout 模型调用。"""
return _run_with_inference_lock(
PIPELINE_LAYOUT_INFERENCE_LOCK, inference_callable, *args, **kwargs
)
def run_mfr_inference(inference_callable, *args, **kwargs):
"""按实验开关执行共享 MFR 模型调用。"""
return _run_with_inference_lock(
PIPELINE_MFR_INFERENCE_LOCK, inference_callable, *args, **kwargs
)
def run_ocr_det_inference(inference_callable, *args, **kwargs):
"""按实验开关执行共享 OCR det 模型调用。"""
return _run_with_inference_lock(
PIPELINE_OCR_DET_INFERENCE_LOCK, inference_callable, *args, **kwargs
)
def run_ocr_rec_inference(inference_callable, *args, **kwargs):
"""按实验开关执行共享 OCR rec 模型调用。"""
return _run_with_inference_lock(
PIPELINE_OCR_REC_INFERENCE_LOCK, inference_callable, *args, **kwargs
)
MFR_MODEL = os.getenv('MINERU_FORMULA_CH_SUPPORT', 'False')
if MFR_MODEL.lower() in ['true', '1', 'yes']:
@@ -1,24 +1,24 @@
# Copyright (c) Opendatalab. All rights reserved.
import copy
import os
from tqdm import tqdm
from mineru.backend.utils.html_image_utils import replace_inline_table_images
from mineru.backend.utils.runtime_utils import cross_page_table_merge
from mineru.utils.config_reader import get_device
from mineru.backend.pipeline.model_init import AtomModelSingleton
from mineru.backend.pipeline.model_init import (
AtomModelSingleton,
run_ocr_rec_inference,
)
from mineru.backend.pipeline.para_split import para_split
from mineru.utils.char_utils import full_to_half
from mineru.utils.cut_image import cut_image_and_table
from mineru.utils.enum_class import ContentType, BlockType
from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info
from mineru.utils.model_utils import clean_memory
from mineru.backend.pipeline.pipeline_magic_model import MagicModel
from mineru.utils.ocr_utils import OcrConfidence, rotate_vertical_crop_if_needed
from mineru.version import __version__
from mineru.utils.hash_utils import bytes_md5
from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard
from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard
def page_model_info_to_page_info(page_model_info, image_dict, page, image_writer, page_index, ocr_enable=False):
@@ -77,20 +77,24 @@ def append_page_model_infos_to_middle_json(
):
for offset, (page_model_info, image_dict) in enumerate(zip(page_model_infos, images_list)):
page_index = page_start_index + offset
with pdfium_guard():
page = pdf_doc[page_index]
page_info = page_model_info_to_page_info(
copy.deepcopy(page_model_info),
image_dict,
page,
image_writer,
page_index,
ocr_enable=ocr_enable,
)
if page_info is None:
page = None
try:
with pdfium_guard():
page_w, page_h = map(int, pdf_doc[page_index].get_size())
page_info = make_page_info_dict([], page_index, page_w, page_h, [])
page = pdf_doc[page_index]
page_info = page_model_info_to_page_info(
copy.deepcopy(page_model_info),
image_dict,
page,
image_writer,
page_index,
ocr_enable=ocr_enable,
)
if page_info is None:
with pdfium_guard():
page_w, page_h = map(int, page.get_size())
page_info = make_page_info_dict([], page_index, page_w, page_h, [])
finally:
close_pdfium_child(page)
middle_json["pdf_info"].append(page_info)
if progress_bar is not None:
progress_bar.update(1)
@@ -231,7 +235,9 @@ def _apply_post_ocr(pdf_info_list, lang=None):
det_db_box_thresh=0.3,
lang=lang
)
ocr_res_list = ocr_model.ocr(img_crop_list, det=False, tqdm_enable=True)[0]
ocr_res_list = run_ocr_rec_inference(
ocr_model.ocr, img_crop_list, det=False, tqdm_enable=True
)[0]
assert len(ocr_res_list) == len(
need_ocr_list), f'ocr_res_list: {len(ocr_res_list)}, need_ocr_list: {len(need_ocr_list)}'
for index, span in enumerate(need_ocr_list):
@@ -280,8 +286,6 @@ def finalize_middle_json(
"""Apply document-level post processing once all page_info entries are ready."""
apply_server_side_postprocess(pdf_info_list, lang=lang)
finalize_middle_json_from_preproc(pdf_info_list)
if os.getenv('MINERU_DONOT_CLEAN_MEM') is None and len(pdf_info_list) >= 10:
clean_memory(get_device())
def init_middle_json():
+82 -76
View File
@@ -168,53 +168,56 @@ def doc_analyze_streaming(
raise ValueError("pdf_bytes_list, image_writer_list, and lang_list must have the same length")
doc_contexts = []
total_pages = 0
for doc_index, (pdf_bytes, image_writer, lang) in enumerate(
zip(pdf_bytes_list, image_writer_list, lang_list)
):
_ocr_enable = _get_ocr_enable(pdf_bytes, parse_method)
pdf_doc = open_pdfium_document(pdfium.PdfDocument, pdf_bytes)
page_count = get_pdfium_document_page_count(pdf_doc)
total_pages += page_count
doc_contexts.append(
{
'doc_index': doc_index,
'pdf_bytes': pdf_bytes,
'pdf_doc': pdf_doc,
'page_count': page_count,
'next_page_idx': 0,
'middle_json': init_middle_json(),
'model_list': [],
'image_writer': image_writer,
'lang': lang,
'ocr_enable': _ocr_enable,
'closed': False,
}
try:
total_pages = 0
for doc_index, (pdf_bytes, image_writer, lang) in enumerate(
zip(pdf_bytes_list, image_writer_list, lang_list)
):
_ocr_enable = _get_ocr_enable(pdf_bytes, parse_method)
pdf_doc = open_pdfium_document(pdfium.PdfDocument, pdf_bytes)
try:
page_count = get_pdfium_document_page_count(pdf_doc)
context = {
'doc_index': doc_index,
'pdf_bytes': pdf_bytes,
'pdf_doc': pdf_doc,
'page_count': page_count,
'next_page_idx': 0,
'middle_json': init_middle_json(),
'model_list': [],
'image_writer': image_writer,
'lang': lang,
'ocr_enable': _ocr_enable,
'closed': False,
}
except Exception:
close_pdfium_document(pdf_doc)
raise
total_pages += page_count
doc_contexts.append(context)
if total_pages == 0:
_emit_zero_page_contexts(
doc_contexts,
on_doc_ready,
client_side_output_generation=client_side_output_generation,
)
return
window_size = get_processing_window_size(default=64)
total_batches = (total_pages + window_size - 1) // window_size
logger.info(
f'Pipeline processing-window multi-file run. doc_count={len(doc_contexts)}, '
f'total_pages={total_pages}, window_size={window_size}, total_batches={total_batches}'
)
if total_pages == 0:
_emit_zero_page_contexts(
doc_contexts,
on_doc_ready,
client_side_output_generation=client_side_output_generation,
)
return
window_size = get_processing_window_size(default=64)
total_batches = (total_pages + window_size - 1) // window_size
logger.info(
f'Pipeline processing-window multi-file run. doc_count={len(doc_contexts)}, '
f'total_pages={total_pages}, window_size={window_size}, total_batches={total_batches}'
)
_emit_zero_page_contexts(
doc_contexts,
on_doc_ready,
client_side_output_generation=client_side_output_generation,
)
processed_pages = 0
infer_start = time.time()
try:
processed_pages = 0
infer_start = time.time()
progress_bar = None
last_append_end_time = None
try:
@@ -264,45 +267,48 @@ def doc_analyze_streaming(
f'batch_pages={len(batch_images)}, doc_slices={_format_doc_slices(batch_slices)}'
)
batch_results = batch_image_analyze(
batch_images,
formula_enable=formula_enable,
table_enable=table_enable,
)
if progress_bar is None:
progress_bar = tqdm(total=total_pages, desc="Processing pages")
else:
exclude_progress_bar_idle_time(
progress_bar,
last_append_end_time,
now=time.time(),
try:
batch_results = batch_image_analyze(
batch_images,
formula_enable=formula_enable,
table_enable=table_enable,
)
result_offset = 0
for context, images_list, page_start, take_count in batch_payloads:
result_slice = batch_results[result_offset: result_offset + take_count]
append_batch_results_to_middle_json(
context['middle_json'],
result_slice,
images_list,
context['pdf_doc'],
context['image_writer'],
page_start_index=page_start,
ocr_enable=context['ocr_enable'],
model_list=context['model_list'],
progress_bar=progress_bar,
)
result_offset += take_count
_close_images(images_list)
images_list.clear()
if context['next_page_idx'] >= context['page_count'] and not context['closed']:
_finalize_processing_window_context(
context,
on_doc_ready,
client_side_output_generation=client_side_output_generation,
if progress_bar is None:
progress_bar = tqdm(total=total_pages, desc="Processing pages")
else:
exclude_progress_bar_idle_time(
progress_bar,
last_append_end_time,
now=time.time(),
)
result_offset = 0
for context, images_list, page_start, take_count in batch_payloads:
result_slice = batch_results[result_offset: result_offset + take_count]
append_batch_results_to_middle_json(
context['middle_json'],
result_slice,
images_list,
context['pdf_doc'],
context['image_writer'],
page_start_index=page_start,
ocr_enable=context['ocr_enable'],
model_list=context['model_list'],
progress_bar=progress_bar,
)
result_offset += take_count
if context['next_page_idx'] >= context['page_count'] and not context['closed']:
_finalize_processing_window_context(
context,
on_doc_ready,
client_side_output_generation=client_side_output_generation,
)
finally:
for _context, images_list, _page_start, _take_count in batch_payloads:
_close_images(images_list)
images_list.clear()
last_append_end_time = time.time()
processed_pages += len(batch_images)
finally:
@@ -16,7 +16,7 @@ from mineru.utils.cut_image import cut_image_and_table
from mineru.utils.enum_class import ContentType
from mineru.utils.hash_utils import bytes_md5
from mineru.utils.title_level_postprocess import apply_title_leveling_to_pdf_info
from mineru.utils.pdfium_guard import close_pdfium_document, pdfium_guard
from mineru.utils.pdfium_guard import close_pdfium_child, close_pdfium_document, pdfium_guard
from mineru.version import __version__
@@ -91,9 +91,13 @@ def append_page_blocks_to_middle_json(
):
for offset, (page_blocks, image_dict) in enumerate(zip(model_output_blocks_list, images_list)):
page_index = page_start_index + offset
with pdfium_guard():
page = pdf_doc[page_index]
page_info = blocks_to_page_info(page_blocks, image_dict, page, image_writer, page_index)
page = None
try:
with pdfium_guard():
page = pdf_doc[page_index]
page_info = blocks_to_page_info(page_blocks, image_dict, page, image_writer, page_index)
finally:
close_pdfium_child(page)
middle_json["pdf_info"].append(page_info)
if progress_bar is not None:
progress_bar.update(1)
+2 -2
View File
@@ -161,7 +161,7 @@ async def parse_request_form(
Form(
description=(
"Defer final markdown/content-list generation to the client. "
"When enabled, the server returns staged middle JSON and images."
"When enabled, the server returns staged middle JSON, model output, and images."
),
),
] = False,
@@ -188,7 +188,7 @@ async def parse_request_form(
if client_side_output_generation:
return_md = False
return_middle_json = True
return_model_output = False
return_model_output = True
return_content_list = False
return_images = True
+9 -5
View File
@@ -1,5 +1,6 @@
# Copyright (c) Opendatalab. All rights reserved.
import asyncio
import multiprocessing
import os
import sys
import threading
@@ -343,8 +344,12 @@ async def mark_task_completed(
def create_visualization_context() -> Optional[VisualizationContext]:
try:
spawn_context = multiprocessing.get_context("spawn")
return VisualizationContext(
executor=ProcessPoolExecutor(max_workers=1),
executor=ProcessPoolExecutor(
max_workers=1,
mp_context=spawn_context,
),
futures=[],
)
except Exception as exc:
@@ -626,9 +631,8 @@ def build_request_form_data(
image_analysis: bool = True,
client_side_output_generation: bool = False,
) -> dict[str, str | list[str]]:
# 开启客户端输出生成时,服务端只负责返回 middle json、图片和原文件。
# 开启客户端输出生成时,只关闭客户端会重建的最终产物。
return_md = not client_side_output_generation
return_model_output = not client_side_output_generation
return_content_list = not client_side_output_generation
return _api_client.build_parse_request_form_data(
lang_list=[lang],
@@ -642,7 +646,7 @@ def build_request_form_data(
end_page_id=end_page_id,
return_md=return_md,
return_middle_json=True,
return_model_output=return_model_output,
return_model_output=True,
return_content_list=return_content_list,
return_images=True,
response_format_zip=True,
@@ -1128,7 +1132,7 @@ async def run_orchestrated_cli(
@click.option(
"--client-side-output-generation",
"client_side_output_generation",
is_flag=True,
type=bool,
default=False,
help=(
"Generate markdown and content lists locally from server-returned "
+43 -2
View File
@@ -23,7 +23,10 @@ from mineru.backend.vlm.vlm_analyze import aio_doc_analyze as aio_vlm_doc_analyz
from mineru.backend.office.pptx_analyze import office_pptx_analyze
from mineru.backend.office.xlsx_analyze import office_xlsx_analyze
from mineru.backend.office.docx_analyze import office_docx_analyze
from mineru.utils.pdfium_guard import rewrite_pdf_bytes_with_pdfium
from mineru.utils.pdfium_guard import (
get_loadable_pdfium_page_indices,
rewrite_pdf_bytes_with_pdfium,
)
os.environ["TORCH_CUDNN_V8_API_DISABLED"] = "1"
if os.getenv("MINERU_LMDEPLOY_DEVICE", "") == "maca":
@@ -190,12 +193,50 @@ def convert_pdf_bytes_to_bytes(pdf_bytes, start_page_id=0, end_page_id=None):
)
if rebuilt_pdf_bytes:
return rebuilt_pdf_bytes
logger.warning("PDFium rewrite returned empty bytes, using original PDF bytes.")
logger.warning(
"PDFium rewrite returned empty bytes, trying to skip broken pages."
)
except Exception as fallback_error:
logger.warning(
f"Error in converting PDF bytes with pdfium: {fallback_error}, "
"trying to skip broken pages."
)
try:
loadable_page_indices, broken_page_indices = get_loadable_pdfium_page_indices(
pdf_bytes,
start_page_id=start_page_id,
end_page_id=end_page_id,
)
if broken_page_indices:
skipped_pages = [page_index + 1 for page_index in broken_page_indices]
logger.warning(
f"Skipped broken PDF pages during PDFium rewrite: {skipped_pages}"
)
if not loadable_page_indices:
logger.warning(
"PDFium skip-broken-page rewrite found no loadable pages, "
"using original PDF bytes."
)
return pdf_bytes
rebuilt_pdf_bytes = rewrite_pdf_bytes_with_pdfium(
pdf_bytes,
start_page_id=start_page_id,
end_page_id=end_page_id,
page_indices=loadable_page_indices,
)
if rebuilt_pdf_bytes:
return rebuilt_pdf_bytes
logger.warning(
"PDFium skip-broken-page rewrite returned empty bytes, "
"using original PDF bytes."
)
except Exception as fallback_error:
logger.warning(
"Error in converting PDF bytes with skip-broken-page fallback: "
f"{fallback_error}, using original PDF bytes."
)
return pdf_bytes
+1 -1
View File
@@ -997,7 +997,7 @@ async def _run_to_markdown_job(
end_page_id=end_pages - 1,
return_md=not use_client_side_output_generation,
return_middle_json=True,
return_model_output=not use_client_side_output_generation,
return_model_output=True,
return_content_list=not use_client_side_output_generation,
return_images=True,
response_format_zip=True,
+1 -1
View File
@@ -112,7 +112,7 @@
</span>
<!-- Code Link. -->
<span class="link-block">
<a href="https://huggingface.co/opendatalab/MinerU2.5-Pro-2604-1.2B" target="_blank" class="external-link button is-normal is-rounded is-dark" style="text-decoration: none; cursor: pointer">
<a href="https://huggingface.co/opendatalab/MinerU2.5-Pro-2605-1.2B" target="_blank" class="external-link button is-normal is-rounded is-dark" style="text-decoration: none; cursor: pointer">
<span class="icon" style="margin-right: 4px">
<i class="fas fa-cube" style="color: white; margin-right: 4px"></i>
</span>
+239 -64
View File
@@ -1,5 +1,6 @@
# Copyright (c) Opendatalab. All rights reserved.
import re
from ctypes import byref, c_int, create_string_buffer
from io import BytesIO
import pypdfium2 as pdfium
@@ -7,6 +8,7 @@ import pypdfium2.raw as pdfium_c
from loguru import logger
from pypdf import PdfReader
from mineru.utils.pdfium_guard import (
close_pdfium_child,
close_pdfium_document,
open_pdfium_document,
pdfium_guard,
@@ -18,6 +20,8 @@ HIGH_IMAGE_COVERAGE_THRESHOLD = 0.8
TEXT_QUALITY_MIN_CHARS = 300
TEXT_QUALITY_BAD_THRESHOLD = 0.03
UNICODE_MAP_ERROR_RATIO_THRESHOLD = 0.04
CID_FONT_USAGE_RATIO_THRESHOLD = 0.01
CID_FONT_USAGE_COUNT_THRESHOLD = 30
MAX_PAGE_ASPECT_RATIO = 10.0
SUSPICIOUS_CJK_72XX_START = 0x7280
SUSPICIOUS_CJK_72XX_END = 0x72DF
@@ -103,7 +107,20 @@ def classify(pdf_bytes):
)
return "ocr"
if detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
cid_font_signal = get_cid_font_signal_pypdf(pdf_bytes, page_indices)
cid_font_usage_signal = _get_cid_font_usage_signal_from_samples(
text_samples,
cid_font_signal,
)
if cid_font_usage_signal["triggered"]:
logger.debug(
"Classify PDF as OCR due to high CID font usage without ToUnicode: "
f"page={cid_font_usage_signal['page_index'] + 1}, "
f"fonts={cid_font_usage_signal['font_names']}, "
f"chars={cid_font_usage_signal['cid_font_char_count']}, "
f"total={cid_font_usage_signal['total_chars']}, "
f"ratio={cid_font_usage_signal['cid_font_usage_ratio']:.4f}"
)
return "ocr"
text_quality_signal = _get_text_quality_signal_from_samples(text_samples)
@@ -196,35 +213,88 @@ def get_extreme_aspect_ratio_page_pdfium(
page_indices,
max_page_aspect_ratio: float = MAX_PAGE_ASPECT_RATIO,
):
for page_index in page_indices:
page = pdf_doc[page_index]
page_width, page_height = page.get_size()
if page_width <= 0 or page_height <= 0:
continue
with pdfium_guard():
for page_index in page_indices:
page = None
try:
page = pdf_doc[page_index]
page_width, page_height = page.get_size()
if page_width <= 0 or page_height <= 0:
continue
aspect_ratio = max(page_width / page_height, page_height / page_width)
if aspect_ratio > max_page_aspect_ratio:
return page_index, aspect_ratio
aspect_ratio = max(page_width / page_height, page_height / page_width)
if aspect_ratio > max_page_aspect_ratio:
return page_index, aspect_ratio
finally:
close_pdfium_child(page)
return None, None
def _collect_pdfium_text_samples(pdf_doc, page_indices):
"""一次性收集抽样页的 textpage 和文本,避免分类链路重复读取 PDFium 文本。"""
text_samples = []
for page_index in page_indices:
page = pdf_doc[page_index]
def _collect_pdfium_text_sample_from_page(page_index, page):
"""从单页 PDFium 对象提取纯 Python 文本统计,并在调用方释放子对象。"""
text_page = None
try:
text_page = page.get_textpage()
text = text_page.get_text_bounded()
text_samples.append(
{
"page_index": page_index,
"text_page": text_page,
"text": text,
"cleaned_text": re.sub(r"\s+", "", text),
}
)
char_count = text_page.count_chars()
null_char_count = 0
replacement_char_count = 0
control_char_count = 0
private_use_char_count = 0
unicode_map_error_count = 0
font_name_counts = {}
for char_index in range(char_count):
unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
if unicode_code == 0:
null_char_count += 1
elif unicode_code == 0xFFFD:
replacement_char_count += 1
elif _is_disallowed_control_unicode(unicode_code):
control_char_count += 1
elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END:
private_use_char_count += 1
if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index):
unicode_map_error_count += 1
font_name = _normalize_pdf_font_name(
_get_pdfium_char_font_name(text_page, char_index)
)
if font_name:
font_name_counts[font_name] = font_name_counts.get(font_name, 0) + 1
return {
"page_index": page_index,
"text": text,
"cleaned_text": re.sub(r"\s+", "", text),
"char_count": char_count,
"null_char_count": null_char_count,
"replacement_char_count": replacement_char_count,
"control_char_count": control_char_count,
"private_use_char_count": private_use_char_count,
"unicode_map_error_count": unicode_map_error_count,
"font_name_counts": font_name_counts,
}
finally:
close_pdfium_child(text_page)
def _collect_pdfium_text_samples(pdf_doc, page_indices):
"""一次性收集抽样页文本统计,返回纯 Python 数据,避免缓存 PDFium 子对象。"""
text_samples = []
with pdfium_guard():
for page_index in page_indices:
page = None
try:
page = pdf_doc[page_index]
text_samples.append(
_collect_pdfium_text_sample_from_page(page_index, page)
)
finally:
close_pdfium_child(page)
return text_samples
@@ -247,7 +317,7 @@ def get_avg_cleaned_chars_per_page_pdfium(pdf_doc, page_indices):
def _get_text_quality_signal_from_samples(text_samples):
"""基于已缓存的 PDFium textpage 统计异常字符质量信号。"""
"""基于已缓存的抽样页字符计数统计异常字符质量信号。"""
total_chars = 0
null_char_count = 0
replacement_char_count = 0
@@ -255,20 +325,11 @@ def _get_text_quality_signal_from_samples(text_samples):
private_use_char_count = 0
for text_sample in text_samples:
text_page = text_sample["text_page"]
char_count = text_page.count_chars()
total_chars += char_count
for char_index in range(char_count):
unicode_code = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
if unicode_code == 0:
null_char_count += 1
elif unicode_code == 0xFFFD:
replacement_char_count += 1
elif _is_disallowed_control_unicode(unicode_code):
control_char_count += 1
elif _PRIVATE_USE_AREA_START <= unicode_code <= _PRIVATE_USE_AREA_END:
private_use_char_count += 1
total_chars += text_sample["char_count"]
null_char_count += text_sample["null_char_count"]
replacement_char_count += text_sample["replacement_char_count"]
control_char_count += text_sample["control_char_count"]
private_use_char_count += text_sample["private_use_char_count"]
abnormal_chars = (
null_char_count
@@ -302,13 +363,8 @@ def _get_unicode_map_error_signal_from_samples(text_samples):
unicode_map_error_count = 0
for text_sample in text_samples:
text_page = text_sample["text_page"]
char_count = text_page.count_chars()
total_chars += char_count
for char_index in range(char_count):
if pdfium_c.FPDFText_HasUnicodeMapError(text_page, char_index):
unicode_map_error_count += 1
total_chars += text_sample["char_count"]
unicode_map_error_count += text_sample["unicode_map_error_count"]
unicode_map_error_ratio = 0.0
if total_chars > 0:
@@ -321,6 +377,102 @@ def _get_unicode_map_error_signal_from_samples(text_samples):
}
def _normalize_pdf_font_name(font_name) -> str:
"""规范化 PDF 字体名,统一 pypdf 的 NameObject 和 PDFium 返回值格式。"""
if font_name is None:
return ""
return str(font_name).strip().lstrip("/")
def _get_pdfium_char_font_name(text_page, char_index: int) -> str:
"""读取 PDFium 字符级字体名,用于统计可疑 CID 字体的实际使用比例。"""
flags = c_int()
buffer_length = pdfium_c.FPDFText_GetFontInfo(
text_page,
char_index,
None,
0,
byref(flags),
)
if buffer_length <= 0:
return ""
font_name_buffer = create_string_buffer(buffer_length)
actual_length = pdfium_c.FPDFText_GetFontInfo(
text_page,
char_index,
font_name_buffer,
buffer_length,
byref(flags),
)
if actual_length <= 0:
return ""
return font_name_buffer.value.decode("utf-8", errors="ignore")
def _get_cid_font_usage_signal_from_samples(text_samples, cid_font_signal):
"""基于 PDFium 字符级字体名统计可疑 CID 字体在抽样页中的真实使用比例。"""
best_signal = {
"triggered": False,
"page_index": None,
"font_names": [],
"cid_font_char_count": 0,
"total_chars": 0,
"cid_font_usage_ratio": 0.0,
}
if not cid_font_signal or not cid_font_signal.get("triggered"):
return best_signal
page_fonts = cid_font_signal.get("page_fonts") or {}
for text_sample in text_samples:
page_index = text_sample.get("page_index")
cid_font_names = {
_normalize_pdf_font_name(font_name)
for font_name in page_fonts.get(page_index, set())
}
cid_font_names.discard("")
if not cid_font_names:
continue
total_chars = text_sample["char_count"]
if total_chars <= 0:
continue
font_name_counts = text_sample.get("font_name_counts") or {}
matched_font_names = cid_font_names.intersection(font_name_counts)
cid_font_char_count = sum(
font_name_counts[font_name] for font_name in matched_font_names
)
cid_font_usage_ratio = cid_font_char_count / total_chars
signal = {
"triggered": False,
"page_index": page_index,
"font_names": sorted(matched_font_names),
"cid_font_char_count": cid_font_char_count,
"total_chars": total_chars,
"cid_font_usage_ratio": cid_font_usage_ratio,
}
if (
cid_font_char_count >= CID_FONT_USAGE_COUNT_THRESHOLD
and cid_font_usage_ratio >= CID_FONT_USAGE_RATIO_THRESHOLD
):
signal["triggered"] = True
return signal
if (
signal["cid_font_usage_ratio"],
signal["cid_font_char_count"],
) > (
best_signal["cid_font_usage_ratio"],
best_signal["cid_font_char_count"],
):
best_signal = signal
return best_signal
def _get_u72xx_text_signal_from_samples(text_samples):
"""基于已缓存的抽样页文本统计扣除常用字后的 U+7280-U+72DF 字符占比。"""
cjk_chars = 0
@@ -435,8 +587,10 @@ def _get_sampled_ascii_punct_signal_from_samples(text_samples):
return best_signal
def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
def get_cid_font_signal_pypdf(pdf_bytes, page_indices):
"""收集抽样页中无 ToUnicode 的 Identity CID 字体资源,供后续按实际字符使用量判定。"""
reader = PdfReader(BytesIO(pdf_bytes))
page_fonts = {}
for page_index in page_indices:
page = reader.pages[page_index]
@@ -448,7 +602,7 @@ def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
if not fonts:
continue
for _, font_ref in fonts.items():
for font_key, font_ref in fonts.items():
font = _resolve_pdf_object(font_ref)
if not font:
continue
@@ -464,9 +618,20 @@ def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
and has_descendant_fonts
and not has_to_unicode
):
return True
font_name = font.get("/BaseFont") or font_key
page_fonts.setdefault(page_index, set()).add(
_normalize_pdf_font_name(font_name)
)
return False
return {
"triggered": bool(page_fonts),
"page_fonts": page_fonts,
}
def detect_cid_font_signal_pypdf(pdf_bytes, page_indices):
"""兼容旧接口:只返回是否存在无 ToUnicode 的 Identity CID 字体资源。"""
return get_cid_font_signal_pypdf(pdf_bytes, page_indices)["triggered"]
def _resolve_pdf_object(obj):
@@ -478,23 +643,33 @@ def _resolve_pdf_object(obj):
def get_high_image_coverage_ratio_pdfium(pdf_doc, page_indices):
high_image_coverage_pages = 0
for page_index in page_indices:
page = pdf_doc[page_index]
page_bbox = page.get_bbox()
page_area = abs(
(page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1])
)
image_area = 0.0
with pdfium_guard():
for page_index in page_indices:
page = None
try:
page = pdf_doc[page_index]
page_bbox = page.get_bbox()
page_area = abs(
(page_bbox[2] - page_bbox[0]) * (page_bbox[3] - page_bbox[1])
)
image_area = 0.0
for page_object in page.get_objects(
filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3
):
left, bottom, right, top = page_object.get_pos()
image_area += max(0.0, right - left) * max(0.0, top - bottom)
for page_object in page.get_objects(
filter=[pdfium_c.FPDF_PAGEOBJ_IMAGE], max_depth=3
):
try:
left, bottom, right, top = page_object.get_pos()
image_area += max(0.0, right - left) * max(0.0, top - bottom)
finally:
close_pdfium_child(page_object)
coverage_ratio = min(image_area / page_area, 1.0) if page_area > 0 else 0.0
if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD:
high_image_coverage_pages += 1
coverage_ratio = (
min(image_area / page_area, 1.0) if page_area > 0 else 0.0
)
if coverage_ratio >= HIGH_IMAGE_COVERAGE_THRESHOLD:
high_image_coverage_pages += 1
finally:
close_pdfium_child(page)
if not page_indices:
return 0.0
+96 -13
View File
@@ -21,6 +21,7 @@ from mineru.utils.enum_class import ImageType
from mineru.utils.hash_utils import str_sha256
from mineru.utils.pdf_page_id import get_end_page_id
from mineru.utils.pdfium_guard import (
close_pdfium_child,
close_pdfium_document,
get_pdfium_document_page_count,
open_pdfium_document,
@@ -35,11 +36,15 @@ DEFAULT_PDF_IMAGE_DPI = 200
# DEFAULT_PDF_IMAGE_DPI = 144
MAX_PDF_RENDER_PROCESSES = 3
MIN_PAGES_PER_RENDER_PROCESS = 30
PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS = 0.1
PDF_RENDER_TERMINATE_GRACE_PERIOD_SECONDS = 0.1
PDF_RENDER_KILL_JOIN_TIMEOUT_SECONDS = 0.1
_pdf_render_executor: ProcessPoolExecutor | None = None
_pdf_render_executor_lock = threading.Lock()
_pdf_render_spawn_submit_lock = threading.Lock()
_pdf_render_spawn_submit_executor_id: int | None = None
_pdf_render_spawn_submit_count = 0
def pdf_page_to_image(
@@ -62,7 +67,10 @@ def pdf_page_to_image(
"scale": scale,
}
if image_type == ImageType.BASE64:
image_dict["img_base64"] = image_to_b64str(pil_img)
try:
image_dict["img_base64"] = image_to_b64str(pil_img)
finally:
pil_img.close()
else:
image_dict["img_pil"] = pil_img
@@ -78,6 +86,18 @@ def _load_images_from_pdf_worker(
)
def _close_image_dicts(images_list) -> None:
"""关闭 image dict 中的 PIL 图片,供异常清理路径释放已生成的图像资源。"""
for image_dict in images_list or []:
pil_img = image_dict.get("img_pil")
if pil_img is None:
continue
try:
pil_img.close()
except Exception:
pass
def _calculate_render_process_count(total_pages: int, threads: int, cpu_count=None) -> int:
requested_threads = max(1, threads)
available_cpus = max(1, cpu_count if cpu_count is not None else (os.cpu_count() or 1))
@@ -137,9 +157,10 @@ def _create_pdf_render_executor(max_workers: int) -> ProcessPoolExecutor:
return ProcessPoolExecutor(max_workers=max_workers)
start_method = multiprocessing.get_start_method()
if start_method == "fork":
if start_method != "spawn":
logger.debug(
"PDF image rendering switches multiprocessing start method from fork to spawn"
"PDF image rendering switches multiprocessing start method "
f"from {start_method} to spawn"
)
return ProcessPoolExecutor(
max_workers=max_workers,
@@ -149,6 +170,44 @@ def _create_pdf_render_executor(max_workers: int) -> ProcessPoolExecutor:
return ProcessPoolExecutor(max_workers=max_workers)
def _is_pdf_render_pool_still_spawning_workers(executor: ProcessPoolExecutor) -> bool:
"""判断渲染进程池是否还可能因为 submit 而继续创建新的 worker。"""
max_workers = getattr(executor, "_max_workers", None)
if max_workers is None or max_workers <= 1:
return False
processes = getattr(executor, "_processes", None)
process_count = 0 if processes is None else len(processes)
return process_count < max_workers
def _submit_pdf_render_task(
executor: ProcessPoolExecutor,
fn,
*args,
**kwargs,
):
"""提交 PDF 渲染任务;冷启动补 worker 时串行 submit 并错开 100ms。"""
global _pdf_render_spawn_submit_executor_id, _pdf_render_spawn_submit_count
with _pdf_render_spawn_submit_lock:
should_throttle = _is_pdf_render_pool_still_spawning_workers(executor)
if should_throttle:
executor_id = id(executor)
if _pdf_render_spawn_submit_executor_id != executor_id:
_pdf_render_spawn_submit_executor_id = executor_id
_pdf_render_spawn_submit_count = 0
elif _pdf_render_spawn_submit_count > 0:
time.sleep(PDF_RENDER_PROCESS_SPAWN_DELAY_SECONDS)
future = executor.submit(fn, *args, **kwargs)
if should_throttle:
_pdf_render_spawn_submit_count += 1
return future
def _get_pdf_render_executor() -> ProcessPoolExecutor:
global _pdf_render_executor
@@ -177,8 +236,14 @@ def _recycle_pdf_render_executor(
_pdf_render_executor = None
if terminate_processes:
_terminate_executor_processes(executor)
executor.shutdown(wait=False, cancel_futures=True)
try:
_terminate_executor_processes(executor)
except Exception as exc:
logger.warning(f"Failed to terminate PDF render executor processes: {exc}")
try:
executor.shutdown(wait=False, cancel_futures=True)
except Exception as exc:
logger.warning(f"Failed to shutdown PDF render executor: {exc}")
def shutdown_pdf_render_executor() -> None:
@@ -228,11 +293,13 @@ def _load_images_from_pdf_bytes_range(
executor = _get_pdf_render_executor()
recycle_executor = False
collected_image_lists = []
try:
futures = []
future_to_range = {}
for range_start, range_end in page_ranges:
future = executor.submit(
future = _submit_pdf_render_task(
executor,
_load_images_from_pdf_worker,
pdf_bytes,
dpi,
@@ -255,6 +322,7 @@ def _load_images_from_pdf_bytes_range(
for future in futures:
range_start = future_to_range[future]
images_list = future.result()
collected_image_lists.append(images_list)
all_results.append((range_start, images_list))
all_results.sort(key=lambda x: x[0])
@@ -262,10 +330,15 @@ def _load_images_from_pdf_bytes_range(
for _, imgs in all_results:
images_list.extend(imgs)
collected_image_lists.clear()
return images_list
except BrokenProcessPool:
recycle_executor = True
raise
except Exception:
for images_list in collected_image_lists:
_close_image_dicts(images_list)
raise
finally:
if recycle_executor:
logger.warning("Recycling persistent PDF render executor after render failure")
@@ -298,7 +371,9 @@ async def aio_load_images_from_pdf_bytes_range(
def _terminate_executor_processes(executor):
"""强制终止 ProcessPoolExecutor 中的所有子进程"""
processes = list(getattr(executor, "_processes", {}).values())
# executor.shutdown() 后 _processes 会被置空,重复回收时直接视为无进程。
process_map = getattr(executor, "_processes", None) or {}
processes = list(process_map.values())
if not processes:
return
@@ -358,9 +433,13 @@ def load_images_from_pdf_core(
for index in range(start_page_id, end_page_id + 1):
# logger.debug(f"Converting page {index}/{pdf_page_num} to image")
page = pdf_doc[index]
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
images_list.append(image_dict)
page = None
try:
page = pdf_doc[index]
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
images_list.append(image_dict)
finally:
close_pdfium_child(page)
finally:
close_pdfium_document(pdf_doc)
@@ -394,9 +473,13 @@ def load_images_from_pdf_doc(
images_list = []
with pdfium_guard():
for index in range(start_page_id, normalized_end_page_id + 1):
page = pdf_doc[index]
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
images_list.append(image_dict)
page = None
try:
page = pdf_doc[index]
image_dict = pdf_page_to_image(page, dpi=dpi, image_type=image_type)
images_list.append(image_dict)
finally:
close_pdfium_child(page)
return images_list
+9 -6
View File
@@ -20,13 +20,16 @@ def page_to_image(
if (long_side_length*scale) > max_width_or_height:
scale = max_width_or_height / long_side_length
bitmap: PdfBitmap = page.render(scale=scale) # type: ignore
image = bitmap.to_pil()
bitmap: PdfBitmap | None = None
try:
bitmap.close()
except Exception as e:
logger.error(f"Failed to close bitmap: {e}")
bitmap = page.render(scale=scale) # type: ignore
image = bitmap.to_pil().copy()
finally:
if bitmap is not None:
try:
bitmap.close()
except Exception as e:
logger.error(f"Failed to close bitmap: {e}")
return image, scale
+22 -17
View File
@@ -6,7 +6,7 @@ import pypdfium2 as pdfium
from pdftext.pdf.chars import deduplicate_chars, get_chars
from pdftext.pdf.pages import assign_scripts, get_blocks, get_lines, get_spans
from mineru.utils.pdfium_guard import pdfium_guard
from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard
def get_page(
@@ -39,25 +39,30 @@ def get_page_chars(
page_char_count: int | None = None,
) -> dict:
"""轻量读取页面字符坐标,供只需要 char 级信息的路径复用。"""
with pdfium_guard():
if textpage is None:
textpage = page.get_textpage()
page_bbox: List[float] = page.get_bbox()
page_width = math.ceil(abs(page_bbox[2] - page_bbox[0]))
page_height = math.ceil(abs(page_bbox[1] - page_bbox[3]))
owns_textpage = textpage is None
try:
with pdfium_guard():
if textpage is None:
textpage = page.get_textpage()
page_bbox: List[float] = page.get_bbox()
page_width = math.ceil(abs(page_bbox[2] - page_bbox[0]))
page_height = math.ceil(abs(page_bbox[1] - page_bbox[3]))
page_rotation = 0
try:
page_rotation = page.get_rotation()
except Exception:
pass
page_rotation = 0
try:
page_rotation = page.get_rotation()
except Exception:
pass
if page_char_count is None:
page_char_count = textpage.count_chars()
if page_char_count is None:
page_char_count = textpage.count_chars()
chars = deduplicate_chars(
get_chars(textpage, page_bbox, page_rotation, quote_loosebox)
)
chars = deduplicate_chars(
get_chars(textpage, page_bbox, page_rotation, quote_loosebox)
)
finally:
if owns_textpage:
close_pdfium_child(textpage)
return {
"bbox": page_bbox,
+71 -4
View File
@@ -4,6 +4,8 @@ from io import BytesIO
from contextlib import contextmanager
from typing import Any, Callable, Sequence, TypeVar
from loguru import logger
from mineru.utils.pdf_page_id import get_end_page_id
@@ -39,6 +41,70 @@ def close_pdfium_document(pdf_doc) -> None:
pdf_doc.close()
def close_pdfium_child(pdfium_obj) -> None:
"""显式关闭 PDFium 子对象,避免依赖 weakref/finalizer 延迟释放 native 资源。"""
if pdfium_obj is None:
return
close = getattr(pdfium_obj, "close", None)
if callable(close):
with pdfium_guard():
close()
def close_pdfium_objects_safely(*pdfium_objs, owner: str = "pdfium cleanup") -> None:
"""清理多个 PDFium 对象时逐个尝试关闭,避免前一个关闭失败阻断后续对象释放。"""
for pdfium_obj in pdfium_objs:
if pdfium_obj is None:
continue
try:
close_pdfium_child(pdfium_obj)
except Exception as exc:
logger.warning(f"Failed to close PDFium object during {owner}: {exc}")
def get_loadable_pdfium_page_indices(
src_pdf_bytes: bytes,
start_page_id: int = 0,
end_page_id: int | None = None,
) -> tuple[list[int], list[int]]:
"""逐页探测 PDFium 可加载页面,返回可保留页和损坏页的 0-based 索引。"""
import pypdfium2 as pdfium
loadable_page_indices = []
broken_page_indices = []
pdf_doc = None
try:
with pdfium_guard():
pdf_doc = pdfium.PdfDocument(src_pdf_bytes)
total_page_count = len(pdf_doc)
if total_page_count == 0:
return [], []
normalized_start_page_id = max(0, start_page_id)
normalized_end_page_id = get_end_page_id(end_page_id, total_page_count)
if normalized_start_page_id > normalized_end_page_id:
return [], []
for page_index in range(
normalized_start_page_id,
normalized_end_page_id + 1,
):
page = None
try:
page = pdf_doc[page_index]
page.get_size()
loadable_page_indices.append(page_index)
except Exception:
broken_page_indices.append(page_index)
finally:
close_pdfium_child(page)
finally:
close_pdfium_document(pdf_doc)
return loadable_page_indices, broken_page_indices
def rewrite_pdf_bytes_with_pdfium(
src_pdf_bytes: bytes,
start_page_id: int = 0,
@@ -79,7 +145,8 @@ def rewrite_pdf_bytes_with_pdfium(
output_doc.save(output_buffer)
return output_buffer.getvalue()
finally:
if output_doc is not None:
close_pdfium_document(output_doc)
if pdf_doc is not None:
close_pdfium_document(pdf_doc)
close_pdfium_objects_safely(
output_doc,
pdf_doc,
owner="rewrite_pdf_bytes_with_pdfium",
)
+77 -74
View File
@@ -12,7 +12,7 @@ from mineru.utils.boxbase import calculate_overlap_area_in_bbox1_area_ratio
from mineru.utils.enum_class import BlockType, ContentType
from mineru.utils.pdf_image_tools import get_crop_img
from mineru.utils.pdf_text_tool import get_lines_from_chars, get_page_chars
from mineru.utils.pdfium_guard import pdfium_guard
from mineru.utils.pdfium_guard import close_pdfium_child, pdfium_guard
MAX_NATIVE_TEXT_CHARS_PER_PAGE = 65535
@@ -35,91 +35,94 @@ def txt_spans_extract(pdf_page, spans, pil_img, scale, all_bboxes, all_discarded
page_char_count = None
textpage = None
try:
with pdfium_guard():
textpage = pdf_page.get_textpage()
page_char_count = textpage.count_chars()
except Exception as exc:
logger.debug(f"Failed to get page char count before txt extraction: {exc}")
try:
with pdfium_guard():
textpage = pdf_page.get_textpage()
page_char_count = textpage.count_chars()
except Exception as exc:
logger.debug(f"Failed to get page char count before txt extraction: {exc}")
if page_char_count is not None and page_char_count > MAX_NATIVE_TEXT_CHARS_PER_PAGE:
logger.info(
"Fallback to post-OCR in txt_spans_extract due to high char count: "
f"count_chars={page_char_count}"
if page_char_count is not None and page_char_count > MAX_NATIVE_TEXT_CHARS_PER_PAGE:
logger.info(
"Fallback to post-OCR in txt_spans_extract due to high char count: "
f"count_chars={page_char_count}"
)
need_ocr_spans = [
span for span in spans if span.get('type') == ContentType.TEXT
]
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
page_chars = get_page_chars(
pdf_page,
textpage=textpage,
page_char_count=page_char_count,
)
need_ocr_spans = [
span for span in spans if span.get('type') == ContentType.TEXT
page_all_chars = [
char for char in page_chars['chars']
if _is_supported_rotation(char['rotation'])
]
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
page_chars = get_page_chars(
pdf_page,
textpage=textpage,
page_char_count=page_char_count,
)
page_all_chars = [
char for char in page_chars['chars']
if _is_supported_rotation(char['rotation'])
]
# 计算所有sapn的高度的中位数
span_height_list = []
for span in spans:
if span['type'] in [ContentType.TEXT]:
span_height = span['bbox'][3] - span['bbox'][1]
span['height'] = span_height
span['width'] = span['bbox'][2] - span['bbox'][0]
span_height_list.append(span_height)
if len(span_height_list) == 0:
return spans
else:
median_span_height = statistics.median(span_height_list)
# 计算所有sapn的高度的中位数
span_height_list = []
for span in spans:
if span['type'] in [ContentType.TEXT]:
span_height = span['bbox'][3] - span['bbox'][1]
span['height'] = span_height
span['width'] = span['bbox'][2] - span['bbox'][0]
span_height_list.append(span_height)
if len(span_height_list) == 0:
return spans
else:
median_span_height = statistics.median(span_height_list)
useful_spans = []
unuseful_spans = []
# 纵向span的两个特征:1. 高度超过多个line 2. 高宽比超过某个值
vertical_spans = []
for span in spans:
if span['type'] in [ContentType.TEXT]:
for block in all_bboxes + all_discarded_blocks:
if block[7] in [BlockType.IMAGE_BODY, BlockType.TABLE_BODY, BlockType.INTERLINE_EQUATION]:
continue
if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5:
if span['height'] > median_span_height * 2.3 and span['height'] > span['width'] * 2.3:
vertical_spans.append(span)
elif block in all_bboxes:
useful_spans.append(span)
else:
unuseful_spans.append(span)
break
useful_spans = []
unuseful_spans = []
# 纵向span的两个特征:1. 高度超过多个line 2. 高宽比超过某个值
vertical_spans = []
for span in spans:
if span['type'] in [ContentType.TEXT]:
for block in all_bboxes + all_discarded_blocks:
if block[7] in [BlockType.IMAGE_BODY, BlockType.TABLE_BODY, BlockType.INTERLINE_EQUATION]:
continue
if calculate_overlap_area_in_bbox1_area_ratio(span['bbox'], block[0:4]) > 0.5:
if span['height'] > median_span_height * 2.3 and span['height'] > span['width'] * 2.3:
vertical_spans.append(span)
elif block in all_bboxes:
useful_spans.append(span)
else:
unuseful_spans.append(span)
break
"""垂直的span框直接用line进行填充"""
if len(vertical_spans) > 0:
page_all_lines = [
line for line in get_lines_from_chars(page_chars['chars'])
if _is_supported_rotation(line['rotation'])
]
for pdfium_line in page_all_lines:
for span in vertical_spans:
if calculate_overlap_area_in_bbox1_area_ratio(pdfium_line['bbox'].bbox, span['bbox']) > 0.5:
for pdfium_span in pdfium_line['spans']:
span['content'] += pdfium_span['text']
break
"""垂直的span框直接用line进行填充"""
if len(vertical_spans) > 0:
page_all_lines = [
line for line in get_lines_from_chars(page_chars['chars'])
if _is_supported_rotation(line['rotation'])
]
for pdfium_line in page_all_lines:
for span in vertical_spans:
if calculate_overlap_area_in_bbox1_area_ratio(pdfium_line['bbox'].bbox, span['bbox']) > 0.5:
for pdfium_span in pdfium_line['spans']:
span['content'] += pdfium_span['text']
break
if len(span['content']) == 0:
spans.remove(span)
for span in vertical_spans:
if len(span['content']) == 0:
spans.remove(span)
"""水平的span框先用char填充,再用ocr填充空的span框"""
new_spans = []
"""水平的span框先用char填充,再用ocr填充空的span框"""
new_spans = []
for span in useful_spans + unuseful_spans:
if span['type'] in [ContentType.TEXT]:
span['chars'] = []
new_spans.append(span)
for span in useful_spans + unuseful_spans:
if span['type'] in [ContentType.TEXT]:
span['chars'] = []
new_spans.append(span)
need_ocr_spans = fill_char_in_spans(new_spans, page_all_chars, median_span_height)
need_ocr_spans = fill_char_in_spans(new_spans, page_all_chars, median_span_height)
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
return _prepare_post_ocr_spans(need_ocr_spans, spans, pil_img, scale)
finally:
close_pdfium_child(textpage)
def _is_supported_rotation(rotation) -> bool:
+12 -1
View File
@@ -1,9 +1,12 @@
# Copyright (c) Opendatalab. All rights reserved.
from __future__ import annotations
import time
from copy import deepcopy
from typing import Any
from loguru import logger
from mineru.utils.config_reader import get_llm_aided_config
from mineru.utils.llm_aided import llm_aided_title
@@ -29,7 +32,15 @@ def _resolve_title_aided_config() -> dict[str, Any] | None:
def apply_title_leveling_to_pdf_info(pdf_info: list[dict[str, Any]]):
title_aided_config = _resolve_title_aided_config()
if title_aided_config:
llm_aided_title(pdf_info, title_aided_config)
start_time = time.perf_counter()
success = False
try:
llm_aided_title(pdf_info, title_aided_config)
success = True
finally:
elapsed = time.perf_counter() - start_time
status = "finished" if success else "failed"
logger.info(f"title leveling {status}, cost: {elapsed:.2f}s")
def finalize_client_side_middle_json(middle_json: dict[str, Any]) -> dict[str, Any]: